adaptive-greedy-search 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- adaptive_greedy_search-2.0.0/LICENSE +21 -0
- adaptive_greedy_search-2.0.0/PKG-INFO +149 -0
- adaptive_greedy_search-2.0.0/README.md +128 -0
- adaptive_greedy_search-2.0.0/adaptive_greedy_search.egg-info/PKG-INFO +149 -0
- adaptive_greedy_search-2.0.0/adaptive_greedy_search.egg-info/SOURCES.txt +12 -0
- adaptive_greedy_search-2.0.0/adaptive_greedy_search.egg-info/dependency_links.txt +1 -0
- adaptive_greedy_search-2.0.0/adaptive_greedy_search.egg-info/requires.txt +2 -0
- adaptive_greedy_search-2.0.0/adaptive_greedy_search.egg-info/top_level.txt +1 -0
- adaptive_greedy_search-2.0.0/ags/__init__.py +20 -0
- adaptive_greedy_search-2.0.0/ags/core.py +404 -0
- adaptive_greedy_search-2.0.0/ags/surrogates.py +91 -0
- adaptive_greedy_search-2.0.0/ags/utils.py +8 -0
- adaptive_greedy_search-2.0.0/pyproject.toml +33 -0
- adaptive_greedy_search-2.0.0/setup.cfg +4 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Mohammad Jawad Hasan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: adaptive-greedy-search
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: AGS (Adaptive Greedy Search): surrogate-guided hyperparameter search with plateau escape and pruned cross-validation.
|
|
5
|
+
Author: Mohammad Jawad Hasan
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/MJwd-Hasan/adaptive-greedy-search
|
|
8
|
+
Project-URL: Repository, https://github.com/MJwd-Hasan/adaptive-greedy-search
|
|
9
|
+
Keywords: hyperparameter-optimization,machine-learning,scikit-learn,AutoML,TPE
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.8
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: numpy>=1.20
|
|
19
|
+
Requires-Dist: scikit-learn>=1.0
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+
# Adaptive Greedy Search (AGS)
|
|
23
|
+
|
|
24
|
+
Surrogate-guided hyperparameter search over a discrete grid. Greedy
|
|
25
|
+
hill-climbing with a TPE (or Gaussian Process) surrogate, radius-based
|
|
26
|
+
plateau escape instead of a full-grid scan, and sequential fold-by-fold
|
|
27
|
+
cross-validation with bound-based pruning so clearly-uncompetitive
|
|
28
|
+
candidates get cut short instead of running to completion.
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from ags import AdaptiveGreedySearch
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Why
|
|
35
|
+
|
|
36
|
+
`GridSearchCV` evaluates every combination — correct, but wasteful once
|
|
37
|
+
the grid gets large. `RandomizedSearchCV` is cheap but has no memory:
|
|
38
|
+
it never uses what it already learned from earlier trials to pick the
|
|
39
|
+
next one. Bayesian approaches like Optuna's TPE fix that by modeling
|
|
40
|
+
which regions of the space look promising, but a general-purpose
|
|
41
|
+
sampler still spends full cross-validation budget on every trial, good
|
|
42
|
+
or bad.
|
|
43
|
+
|
|
44
|
+
AGS combines both ideas for the specific case of a **discrete**
|
|
45
|
+
hyperparameter grid:
|
|
46
|
+
|
|
47
|
+
- **A surrogate model** (TPE by default) learns from every evaluated
|
|
48
|
+
point which regions of the grid look promising, and steers the next
|
|
49
|
+
choice there instead of sampling blindly.
|
|
50
|
+
- **Greedy hill-climbing with plateau escape**: it moves to the best
|
|
51
|
+
neighboring grid point each step, and when the immediate neighborhood
|
|
52
|
+
is exhausted, expands outward ring by ring (bounded by
|
|
53
|
+
`max_neighbor_radius`) rather than falling back to scanning the whole
|
|
54
|
+
grid.
|
|
55
|
+
- **Pruned cross-validation**: each candidate's CV folds run
|
|
56
|
+
sequentially, and a candidate is stopped early once it's
|
|
57
|
+
mathematically or statistically out of contention against the best
|
|
58
|
+
result seen so far — so compute isn't wasted finishing out folds for
|
|
59
|
+
a configuration that's already lost.
|
|
60
|
+
|
|
61
|
+
The result: for a fixed evaluation budget, it tends to land on a better
|
|
62
|
+
configuration than random search on reasonably smooth hyperparameter
|
|
63
|
+
landscapes, and at lower wall-clock cost than grid search or plain
|
|
64
|
+
Bayesian search, because it also cuts the *inside* of each evaluation
|
|
65
|
+
(the CV folds), not just the *number* of evaluations.
|
|
66
|
+
|
|
67
|
+
**Where it's not the right tool:** landscapes where quality is scattered
|
|
68
|
+
with no local structure (grid distance doesn't correlate with
|
|
69
|
+
performance), very small evaluation budgets (the surrogate needs a
|
|
70
|
+
handful of points before it's useful), or cases where you need the
|
|
71
|
+
completeness guarantee of exhaustive grid search.
|
|
72
|
+
|
|
73
|
+
## Installation
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
pip install adaptive-greedy-search
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Or, from a cloned copy of this repository:
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
pip install -e .
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Requires Python 3.8+, `numpy`, and `scikit-learn` (installed automatically).
|
|
86
|
+
|
|
87
|
+
## Quickstart
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from ags import AdaptiveGreedySearch
|
|
91
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
92
|
+
from sklearn.datasets import make_classification
|
|
93
|
+
|
|
94
|
+
X, y = make_classification(n_samples=500, n_features=20, random_state=0)
|
|
95
|
+
|
|
96
|
+
param_grid = {
|
|
97
|
+
"n_estimators": [50, 100, 150, 200],
|
|
98
|
+
"max_depth": [3, 5, 7, 9, None],
|
|
99
|
+
"min_samples_split": [2, 4, 6, 8],
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
search = AdaptiveGreedySearch(
|
|
103
|
+
RandomForestClassifier(random_state=0),
|
|
104
|
+
param_grid,
|
|
105
|
+
cv=5, # pruning gives little benefit below cv=5; use 5-10
|
|
106
|
+
scoring="accuracy",
|
|
107
|
+
max_evaluations=30,
|
|
108
|
+
)
|
|
109
|
+
search.fit(X, y)
|
|
110
|
+
|
|
111
|
+
print(search.best_params)
|
|
112
|
+
print(search.best_score)
|
|
113
|
+
print(f"folds saved by pruning: {search.folds_saved}/{search.total_folds_possible}")
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## Key parameters
|
|
117
|
+
|
|
118
|
+
| Parameter | Default | What it does |
|
|
119
|
+
|---|---|---|
|
|
120
|
+
| `cv` | `5` | Number of CV folds. Pruning is far more effective at `5` or `10`. |
|
|
121
|
+
| `scoring` | `"accuracy"` | Any scikit-learn scorer string (`"accuracy"`, `"neg_mean_squared_error"`, `"roc_auc"`, ...). |
|
|
122
|
+
| `max_evaluations` | `25` | Hard cap on the number of grid points evaluated. |
|
|
123
|
+
| `surrogate_type` | `"tpe"` | `"tpe"` (Tree-structured Parzen Estimator, cheap to refit) or `"gp"` (Gaussian Process, better calibrated uncertainty but costlier as evaluations grow). |
|
|
124
|
+
| `pruning_strategy` | `"percentile"` | `"percentile"` (aggressive — compares a candidate against the historical distribution of other candidates at the same fold count), `"optimistic"` (safe — only prunes when a candidate is *mathematically* unable to beat the current best), or `"none"` (exhaustive CV, no pruning). |
|
|
125
|
+
| `pruning_percentile` | `25` | Lower = stricter pruning (fewer candidates survive) when using `"percentile"`. |
|
|
126
|
+
| `early_stopping_patience` | `5` | Stop the whole search if the best score hasn't improved for this many consecutive evaluations. Set to `None` to always run to `max_evaluations`. |
|
|
127
|
+
| `max_neighbor_radius` | `4` | How many rings outward the plateau-escape step is allowed to search before falling back to scoring the remaining grid. |
|
|
128
|
+
|
|
129
|
+
## Choosing a pruning strategy
|
|
130
|
+
|
|
131
|
+
- Want speed and can tolerate a small chance of pruning a candidate
|
|
132
|
+
that might have recovered? Use the default `"percentile"`.
|
|
133
|
+
- Want a guarantee that pruning never changes the final answer versus
|
|
134
|
+
running exhaustively? Use `"optimistic"`.
|
|
135
|
+
- Establishing a baseline, or debugging unexpected results? Use
|
|
136
|
+
`"none"` to fall back to plain exhaustive cross-validation.
|
|
137
|
+
|
|
138
|
+
## What you get back after `.fit(X, y)`
|
|
139
|
+
|
|
140
|
+
- `best_params`, `best_score`, `best_state` — the winning configuration.
|
|
141
|
+
- `history` — a list of dicts, one per evaluated candidate, including
|
|
142
|
+
`n_folds_used`, `n_folds_total`, and whether/why it was pruned.
|
|
143
|
+
- `n_evaluations`, `total_time`, `stopped_early`.
|
|
144
|
+
- `n_pruned`, `total_folds_run`, `total_folds_possible`, `folds_saved`
|
|
145
|
+
— how much cross-validation work was actually skipped.
|
|
146
|
+
|
|
147
|
+
## License
|
|
148
|
+
|
|
149
|
+
MIT © Mohammad Jawad Hasan
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
# Adaptive Greedy Search (AGS)
|
|
2
|
+
|
|
3
|
+
Surrogate-guided hyperparameter search over a discrete grid. Greedy
|
|
4
|
+
hill-climbing with a TPE (or Gaussian Process) surrogate, radius-based
|
|
5
|
+
plateau escape instead of a full-grid scan, and sequential fold-by-fold
|
|
6
|
+
cross-validation with bound-based pruning so clearly-uncompetitive
|
|
7
|
+
candidates get cut short instead of running to completion.
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from ags import AdaptiveGreedySearch
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Why
|
|
14
|
+
|
|
15
|
+
`GridSearchCV` evaluates every combination — correct, but wasteful once
|
|
16
|
+
the grid gets large. `RandomizedSearchCV` is cheap but has no memory:
|
|
17
|
+
it never uses what it already learned from earlier trials to pick the
|
|
18
|
+
next one. Bayesian approaches like Optuna's TPE fix that by modeling
|
|
19
|
+
which regions of the space look promising, but a general-purpose
|
|
20
|
+
sampler still spends full cross-validation budget on every trial, good
|
|
21
|
+
or bad.
|
|
22
|
+
|
|
23
|
+
AGS combines both ideas for the specific case of a **discrete**
|
|
24
|
+
hyperparameter grid:
|
|
25
|
+
|
|
26
|
+
- **A surrogate model** (TPE by default) learns from every evaluated
|
|
27
|
+
point which regions of the grid look promising, and steers the next
|
|
28
|
+
choice there instead of sampling blindly.
|
|
29
|
+
- **Greedy hill-climbing with plateau escape**: it moves to the best
|
|
30
|
+
neighboring grid point each step, and when the immediate neighborhood
|
|
31
|
+
is exhausted, expands outward ring by ring (bounded by
|
|
32
|
+
`max_neighbor_radius`) rather than falling back to scanning the whole
|
|
33
|
+
grid.
|
|
34
|
+
- **Pruned cross-validation**: each candidate's CV folds run
|
|
35
|
+
sequentially, and a candidate is stopped early once it's
|
|
36
|
+
mathematically or statistically out of contention against the best
|
|
37
|
+
result seen so far — so compute isn't wasted finishing out folds for
|
|
38
|
+
a configuration that's already lost.
|
|
39
|
+
|
|
40
|
+
The result: for a fixed evaluation budget, it tends to land on a better
|
|
41
|
+
configuration than random search on reasonably smooth hyperparameter
|
|
42
|
+
landscapes, and at lower wall-clock cost than grid search or plain
|
|
43
|
+
Bayesian search, because it also cuts the *inside* of each evaluation
|
|
44
|
+
(the CV folds), not just the *number* of evaluations.
|
|
45
|
+
|
|
46
|
+
**Where it's not the right tool:** landscapes where quality is scattered
|
|
47
|
+
with no local structure (grid distance doesn't correlate with
|
|
48
|
+
performance), very small evaluation budgets (the surrogate needs a
|
|
49
|
+
handful of points before it's useful), or cases where you need the
|
|
50
|
+
completeness guarantee of exhaustive grid search.
|
|
51
|
+
|
|
52
|
+
## Installation
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install adaptive-greedy-search
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Or, from a cloned copy of this repository:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install -e .
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Requires Python 3.8+, `numpy`, and `scikit-learn` (installed automatically).
|
|
65
|
+
|
|
66
|
+
## Quickstart
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from ags import AdaptiveGreedySearch
|
|
70
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
71
|
+
from sklearn.datasets import make_classification
|
|
72
|
+
|
|
73
|
+
X, y = make_classification(n_samples=500, n_features=20, random_state=0)
|
|
74
|
+
|
|
75
|
+
param_grid = {
|
|
76
|
+
"n_estimators": [50, 100, 150, 200],
|
|
77
|
+
"max_depth": [3, 5, 7, 9, None],
|
|
78
|
+
"min_samples_split": [2, 4, 6, 8],
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
search = AdaptiveGreedySearch(
|
|
82
|
+
RandomForestClassifier(random_state=0),
|
|
83
|
+
param_grid,
|
|
84
|
+
cv=5, # pruning gives little benefit below cv=5; use 5-10
|
|
85
|
+
scoring="accuracy",
|
|
86
|
+
max_evaluations=30,
|
|
87
|
+
)
|
|
88
|
+
search.fit(X, y)
|
|
89
|
+
|
|
90
|
+
print(search.best_params)
|
|
91
|
+
print(search.best_score)
|
|
92
|
+
print(f"folds saved by pruning: {search.folds_saved}/{search.total_folds_possible}")
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Key parameters
|
|
96
|
+
|
|
97
|
+
| Parameter | Default | What it does |
|
|
98
|
+
|---|---|---|
|
|
99
|
+
| `cv` | `5` | Number of CV folds. Pruning is far more effective at `5` or `10`. |
|
|
100
|
+
| `scoring` | `"accuracy"` | Any scikit-learn scorer string (`"accuracy"`, `"neg_mean_squared_error"`, `"roc_auc"`, ...). |
|
|
101
|
+
| `max_evaluations` | `25` | Hard cap on the number of grid points evaluated. |
|
|
102
|
+
| `surrogate_type` | `"tpe"` | `"tpe"` (Tree-structured Parzen Estimator, cheap to refit) or `"gp"` (Gaussian Process, better calibrated uncertainty but costlier as evaluations grow). |
|
|
103
|
+
| `pruning_strategy` | `"percentile"` | `"percentile"` (aggressive — compares a candidate against the historical distribution of other candidates at the same fold count), `"optimistic"` (safe — only prunes when a candidate is *mathematically* unable to beat the current best), or `"none"` (exhaustive CV, no pruning). |
|
|
104
|
+
| `pruning_percentile` | `25` | Lower = stricter pruning (fewer candidates survive) when using `"percentile"`. |
|
|
105
|
+
| `early_stopping_patience` | `5` | Stop the whole search if the best score hasn't improved for this many consecutive evaluations. Set to `None` to always run to `max_evaluations`. |
|
|
106
|
+
| `max_neighbor_radius` | `4` | How many rings outward the plateau-escape step is allowed to search before falling back to scoring the remaining grid. |
|
|
107
|
+
|
|
108
|
+
## Choosing a pruning strategy
|
|
109
|
+
|
|
110
|
+
- Want speed and can tolerate a small chance of pruning a candidate
|
|
111
|
+
that might have recovered? Use the default `"percentile"`.
|
|
112
|
+
- Want a guarantee that pruning never changes the final answer versus
|
|
113
|
+
running exhaustively? Use `"optimistic"`.
|
|
114
|
+
- Establishing a baseline, or debugging unexpected results? Use
|
|
115
|
+
`"none"` to fall back to plain exhaustive cross-validation.
|
|
116
|
+
|
|
117
|
+
## What you get back after `.fit(X, y)`
|
|
118
|
+
|
|
119
|
+
- `best_params`, `best_score`, `best_state` — the winning configuration.
|
|
120
|
+
- `history` — a list of dicts, one per evaluated candidate, including
|
|
121
|
+
`n_folds_used`, `n_folds_total`, and whether/why it was pruned.
|
|
122
|
+
- `n_evaluations`, `total_time`, `stopped_early`.
|
|
123
|
+
- `n_pruned`, `total_folds_run`, `total_folds_possible`, `folds_saved`
|
|
124
|
+
— how much cross-validation work was actually skipped.
|
|
125
|
+
|
|
126
|
+
## License
|
|
127
|
+
|
|
128
|
+
MIT © Mohammad Jawad Hasan
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: adaptive-greedy-search
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: AGS (Adaptive Greedy Search): surrogate-guided hyperparameter search with plateau escape and pruned cross-validation.
|
|
5
|
+
Author: Mohammad Jawad Hasan
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/MJwd-Hasan/adaptive-greedy-search
|
|
8
|
+
Project-URL: Repository, https://github.com/MJwd-Hasan/adaptive-greedy-search
|
|
9
|
+
Keywords: hyperparameter-optimization,machine-learning,scikit-learn,AutoML,TPE
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.8
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: numpy>=1.20
|
|
19
|
+
Requires-Dist: scikit-learn>=1.0
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+
# Adaptive Greedy Search (AGS)
|
|
23
|
+
|
|
24
|
+
Surrogate-guided hyperparameter search over a discrete grid. Greedy
|
|
25
|
+
hill-climbing with a TPE (or Gaussian Process) surrogate, radius-based
|
|
26
|
+
plateau escape instead of a full-grid scan, and sequential fold-by-fold
|
|
27
|
+
cross-validation with bound-based pruning so clearly-uncompetitive
|
|
28
|
+
candidates get cut short instead of running to completion.
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from ags import AdaptiveGreedySearch
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Why
|
|
35
|
+
|
|
36
|
+
`GridSearchCV` evaluates every combination — correct, but wasteful once
|
|
37
|
+
the grid gets large. `RandomizedSearchCV` is cheap but has no memory:
|
|
38
|
+
it never uses what it already learned from earlier trials to pick the
|
|
39
|
+
next one. Bayesian approaches like Optuna's TPE fix that by modeling
|
|
40
|
+
which regions of the space look promising, but a general-purpose
|
|
41
|
+
sampler still spends full cross-validation budget on every trial, good
|
|
42
|
+
or bad.
|
|
43
|
+
|
|
44
|
+
AGS combines both ideas for the specific case of a **discrete**
|
|
45
|
+
hyperparameter grid:
|
|
46
|
+
|
|
47
|
+
- **A surrogate model** (TPE by default) learns from every evaluated
|
|
48
|
+
point which regions of the grid look promising, and steers the next
|
|
49
|
+
choice there instead of sampling blindly.
|
|
50
|
+
- **Greedy hill-climbing with plateau escape**: it moves to the best
|
|
51
|
+
neighboring grid point each step, and when the immediate neighborhood
|
|
52
|
+
is exhausted, expands outward ring by ring (bounded by
|
|
53
|
+
`max_neighbor_radius`) rather than falling back to scanning the whole
|
|
54
|
+
grid.
|
|
55
|
+
- **Pruned cross-validation**: each candidate's CV folds run
|
|
56
|
+
sequentially, and a candidate is stopped early once it's
|
|
57
|
+
mathematically or statistically out of contention against the best
|
|
58
|
+
result seen so far — so compute isn't wasted finishing out folds for
|
|
59
|
+
a configuration that's already lost.
|
|
60
|
+
|
|
61
|
+
The result: for a fixed evaluation budget, it tends to land on a better
|
|
62
|
+
configuration than random search on reasonably smooth hyperparameter
|
|
63
|
+
landscapes, and at lower wall-clock cost than grid search or plain
|
|
64
|
+
Bayesian search, because it also cuts the *inside* of each evaluation
|
|
65
|
+
(the CV folds), not just the *number* of evaluations.
|
|
66
|
+
|
|
67
|
+
**Where it's not the right tool:** landscapes where quality is scattered
|
|
68
|
+
with no local structure (grid distance doesn't correlate with
|
|
69
|
+
performance), very small evaluation budgets (the surrogate needs a
|
|
70
|
+
handful of points before it's useful), or cases where you need the
|
|
71
|
+
completeness guarantee of exhaustive grid search.
|
|
72
|
+
|
|
73
|
+
## Installation
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
pip install adaptive-greedy-search
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Or, from a cloned copy of this repository:
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
pip install -e .
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Requires Python 3.8+, `numpy`, and `scikit-learn` (installed automatically).
|
|
86
|
+
|
|
87
|
+
## Quickstart
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from ags import AdaptiveGreedySearch
|
|
91
|
+
from sklearn.ensemble import RandomForestClassifier
|
|
92
|
+
from sklearn.datasets import make_classification
|
|
93
|
+
|
|
94
|
+
X, y = make_classification(n_samples=500, n_features=20, random_state=0)
|
|
95
|
+
|
|
96
|
+
param_grid = {
|
|
97
|
+
"n_estimators": [50, 100, 150, 200],
|
|
98
|
+
"max_depth": [3, 5, 7, 9, None],
|
|
99
|
+
"min_samples_split": [2, 4, 6, 8],
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
search = AdaptiveGreedySearch(
|
|
103
|
+
RandomForestClassifier(random_state=0),
|
|
104
|
+
param_grid,
|
|
105
|
+
cv=5, # pruning gives little benefit below cv=5; use 5-10
|
|
106
|
+
scoring="accuracy",
|
|
107
|
+
max_evaluations=30,
|
|
108
|
+
)
|
|
109
|
+
search.fit(X, y)
|
|
110
|
+
|
|
111
|
+
print(search.best_params)
|
|
112
|
+
print(search.best_score)
|
|
113
|
+
print(f"folds saved by pruning: {search.folds_saved}/{search.total_folds_possible}")
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## Key parameters
|
|
117
|
+
|
|
118
|
+
| Parameter | Default | What it does |
|
|
119
|
+
|---|---|---|
|
|
120
|
+
| `cv` | `5` | Number of CV folds. Pruning is far more effective at `5` or `10`. |
|
|
121
|
+
| `scoring` | `"accuracy"` | Any scikit-learn scorer string (`"accuracy"`, `"neg_mean_squared_error"`, `"roc_auc"`, ...). |
|
|
122
|
+
| `max_evaluations` | `25` | Hard cap on the number of grid points evaluated. |
|
|
123
|
+
| `surrogate_type` | `"tpe"` | `"tpe"` (Tree-structured Parzen Estimator, cheap to refit) or `"gp"` (Gaussian Process, better calibrated uncertainty but costlier as evaluations grow). |
|
|
124
|
+
| `pruning_strategy` | `"percentile"` | `"percentile"` (aggressive — compares a candidate against the historical distribution of other candidates at the same fold count), `"optimistic"` (safe — only prunes when a candidate is *mathematically* unable to beat the current best), or `"none"` (exhaustive CV, no pruning). |
|
|
125
|
+
| `pruning_percentile` | `25` | Lower = stricter pruning (fewer candidates survive) when using `"percentile"`. |
|
|
126
|
+
| `early_stopping_patience` | `5` | Stop the whole search if the best score hasn't improved for this many consecutive evaluations. Set to `None` to always run to `max_evaluations`. |
|
|
127
|
+
| `max_neighbor_radius` | `4` | How many rings outward the plateau-escape step is allowed to search before falling back to scoring the remaining grid. |
|
|
128
|
+
|
|
129
|
+
## Choosing a pruning strategy
|
|
130
|
+
|
|
131
|
+
- Want speed and can tolerate a small chance of pruning a candidate
|
|
132
|
+
that might have recovered? Use the default `"percentile"`.
|
|
133
|
+
- Want a guarantee that pruning never changes the final answer versus
|
|
134
|
+
running exhaustively? Use `"optimistic"`.
|
|
135
|
+
- Establishing a baseline, or debugging unexpected results? Use
|
|
136
|
+
`"none"` to fall back to plain exhaustive cross-validation.
|
|
137
|
+
|
|
138
|
+
## What you get back after `.fit(X, y)`
|
|
139
|
+
|
|
140
|
+
- `best_params`, `best_score`, `best_state` — the winning configuration.
|
|
141
|
+
- `history` — a list of dicts, one per evaluated candidate, including
|
|
142
|
+
`n_folds_used`, `n_folds_total`, and whether/why it was pruned.
|
|
143
|
+
- `n_evaluations`, `total_time`, `stopped_early`.
|
|
144
|
+
- `n_pruned`, `total_folds_run`, `total_folds_possible`, `folds_saved`
|
|
145
|
+
— how much cross-validation work was actually skipped.
|
|
146
|
+
|
|
147
|
+
## License
|
|
148
|
+
|
|
149
|
+
MIT © Mohammad Jawad Hasan
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
adaptive_greedy_search.egg-info/PKG-INFO
|
|
5
|
+
adaptive_greedy_search.egg-info/SOURCES.txt
|
|
6
|
+
adaptive_greedy_search.egg-info/dependency_links.txt
|
|
7
|
+
adaptive_greedy_search.egg-info/requires.txt
|
|
8
|
+
adaptive_greedy_search.egg-info/top_level.txt
|
|
9
|
+
ags/__init__.py
|
|
10
|
+
ags/core.py
|
|
11
|
+
ags/surrogates.py
|
|
12
|
+
ags/utils.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ags
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""
|
|
2
|
+
AGS -- Adaptive Greedy Search
|
|
3
|
+
|
|
4
|
+
Surrogate-guided hyperparameter search over a discrete grid: greedy
|
|
5
|
+
radius-1 hill-climbing with a TPE (or GP) surrogate, radius-based
|
|
6
|
+
plateau escape instead of a full-grid scan, and sequential fold-by-fold
|
|
7
|
+
cross-validation with bound-based pruning to skip clearly-uncompetitive
|
|
8
|
+
candidates early.
|
|
9
|
+
|
|
10
|
+
from ags import AdaptiveGreedySearch
|
|
11
|
+
|
|
12
|
+
is all you need -- everything else (numpy, scikit-learn) is imported
|
|
13
|
+
internally by the package.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from .core import AdaptiveGreedySearch
|
|
17
|
+
from .surrogates import TPESurrogate, build_gp_surrogate
|
|
18
|
+
|
|
19
|
+
__all__ = ["AdaptiveGreedySearch", "TPESurrogate", "build_gp_surrogate"]
|
|
20
|
+
__version__ = "2.0.0"
|
|
@@ -0,0 +1,404 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Core AdaptiveGreedySearch implementation (AGS v2).
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import itertools
|
|
6
|
+
import time
|
|
7
|
+
import numpy as np
|
|
8
|
+
|
|
9
|
+
from sklearn.base import clone, is_classifier
|
|
10
|
+
from sklearn.model_selection import KFold, StratifiedKFold
|
|
11
|
+
from sklearn.metrics import get_scorer
|
|
12
|
+
|
|
13
|
+
from .surrogates import make_surrogate
|
|
14
|
+
from .utils import safe_index
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class AdaptiveGreedySearch:
|
|
18
|
+
"""
|
|
19
|
+
Surrogate-guided greedy search over a discrete hyperparameter grid.
|
|
20
|
+
|
|
21
|
+
Search strategy:
|
|
22
|
+
1. Random initial exploration.
|
|
23
|
+
2. Surrogate model fitting (TPE by default, or GP).
|
|
24
|
+
3. Greedy radius-1 neighbor selection (batched surrogate UCB).
|
|
25
|
+
4. Radius-based plateau escape when the immediate neighborhood
|
|
26
|
+
is exhausted (expands outward, stops at the first ring with
|
|
27
|
+
unevaluated candidates, scores them in one batched call).
|
|
28
|
+
5. Batched surrogate fallback over any remaining states, for the
|
|
29
|
+
rare case where even the expanded neighborhood is exhausted.
|
|
30
|
+
6. Optional early stopping if the best score hasn't improved for
|
|
31
|
+
`early_stopping_patience` consecutive evaluations.
|
|
32
|
+
|
|
33
|
+
Evaluation:
|
|
34
|
+
Each candidate is scored via cross-validation, run SEQUENTIALLY
|
|
35
|
+
fold by fold (not parallel `cross_val_score`), so that a
|
|
36
|
+
bound-based pruning rule can stop a clearly-uncompetitive
|
|
37
|
+
candidate early instead of finishing out every fold:
|
|
38
|
+
- "optimistic": safe/conservative. Prunes only when even the
|
|
39
|
+
best possible outcome for the remaining folds (assuming
|
|
40
|
+
they all hit the scorer's theoretical max) can't beat the
|
|
41
|
+
current best.
|
|
42
|
+
- "percentile": more aggressive. Prunes when a candidate's
|
|
43
|
+
partial mean falls below a percentile of how other
|
|
44
|
+
candidates were doing at the same fold count.
|
|
45
|
+
|
|
46
|
+
Acquisition:
|
|
47
|
+
UCB = predicted_mean + exploration_weight * predicted_std
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(
|
|
51
|
+
self,
|
|
52
|
+
estimator,
|
|
53
|
+
param_grid,
|
|
54
|
+
cv=5,
|
|
55
|
+
scoring="accuracy",
|
|
56
|
+
initial_points=8,
|
|
57
|
+
max_evaluations=25,
|
|
58
|
+
exploration_weight=2.0,
|
|
59
|
+
random_state=42,
|
|
60
|
+
max_neighbor_radius=4,
|
|
61
|
+
early_stopping_patience=5,
|
|
62
|
+
surrogate_type="tpe",
|
|
63
|
+
tpe_gamma=0.25,
|
|
64
|
+
tpe_bandwidth=1.0,
|
|
65
|
+
enable_pruning=True,
|
|
66
|
+
pruning_strategy="percentile", # "optimistic" | "percentile" | "none"
|
|
67
|
+
pruning_margin=0.0,
|
|
68
|
+
pruning_percentile=25,
|
|
69
|
+
min_folds_before_pruning=2,
|
|
70
|
+
min_history_for_percentile=10,
|
|
71
|
+
score_upper_bound=None,
|
|
72
|
+
):
|
|
73
|
+
|
|
74
|
+
self.estimator = estimator
|
|
75
|
+
self.param_grid = param_grid
|
|
76
|
+
self.cv = cv
|
|
77
|
+
self.scoring = scoring
|
|
78
|
+
|
|
79
|
+
self.initial_points = initial_points
|
|
80
|
+
self.max_evaluations = max_evaluations
|
|
81
|
+
self.exploration_weight = exploration_weight
|
|
82
|
+
self.random_state = random_state
|
|
83
|
+
self.max_neighbor_radius = max_neighbor_radius
|
|
84
|
+
self.early_stopping_patience = early_stopping_patience
|
|
85
|
+
self.surrogate_type = surrogate_type
|
|
86
|
+
self.tpe_gamma = tpe_gamma
|
|
87
|
+
self.tpe_bandwidth = tpe_bandwidth
|
|
88
|
+
|
|
89
|
+
self.enable_pruning = enable_pruning
|
|
90
|
+
self.pruning_strategy = pruning_strategy
|
|
91
|
+
self.pruning_margin = pruning_margin
|
|
92
|
+
self.pruning_percentile = pruning_percentile
|
|
93
|
+
self.min_folds_before_pruning = min_folds_before_pruning
|
|
94
|
+
self.min_history_for_percentile = min_history_for_percentile
|
|
95
|
+
|
|
96
|
+
self.rng = np.random.default_rng(random_state)
|
|
97
|
+
|
|
98
|
+
self.parameter_names = list(param_grid.keys())
|
|
99
|
+
self.values = [list(param_grid[p]) for p in self.parameter_names]
|
|
100
|
+
self.all_states = list(
|
|
101
|
+
itertools.product(*[range(len(v)) for v in self.values])
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
self.evaluated = {}
|
|
105
|
+
self.history = []
|
|
106
|
+
|
|
107
|
+
# --- pruning bookkeeping ---
|
|
108
|
+
self._partial_mean_history = {} # {fold_index: [partial_means...]}
|
|
109
|
+
self._observed_max_fold_score = -np.inf
|
|
110
|
+
self.score_upper_bound = (
|
|
111
|
+
score_upper_bound
|
|
112
|
+
if score_upper_bound is not None
|
|
113
|
+
else self._infer_score_upper_bound(scoring)
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
if is_classifier(estimator):
|
|
117
|
+
self._cv_splitter = StratifiedKFold(
|
|
118
|
+
n_splits=cv, shuffle=True, random_state=random_state
|
|
119
|
+
)
|
|
120
|
+
else:
|
|
121
|
+
self._cv_splitter = KFold(
|
|
122
|
+
n_splits=cv, shuffle=True, random_state=random_state
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
self._scorer = get_scorer(scoring)
|
|
126
|
+
|
|
127
|
+
self.surrogate = make_surrogate(
|
|
128
|
+
surrogate_type,
|
|
129
|
+
tpe_gamma=tpe_gamma,
|
|
130
|
+
tpe_bandwidth=tpe_bandwidth,
|
|
131
|
+
random_state=random_state,
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
# ========================================================
|
|
135
|
+
# Upper bound inference for the "optimistic" pruning rule.
|
|
136
|
+
# ========================================================
|
|
137
|
+
|
|
138
|
+
@staticmethod
|
|
139
|
+
def _infer_score_upper_bound(scoring):
|
|
140
|
+
if not isinstance(scoring, str):
|
|
141
|
+
return None
|
|
142
|
+
if scoring.startswith("neg_"):
|
|
143
|
+
return 0.0 # e.g. neg_mean_squared_error: perfect = 0
|
|
144
|
+
return 1.0 # accuracy, f1*, roc_auc*, r2, precision*, recall*, etc.
|
|
145
|
+
|
|
146
|
+
# ========================================================
|
|
147
|
+
# STATE -> REAL HYPERPARAMETERS
|
|
148
|
+
# ========================================================
|
|
149
|
+
|
|
150
|
+
def state_to_params(self, state):
|
|
151
|
+
return {
|
|
152
|
+
p: self.values[j][state[j]]
|
|
153
|
+
for j, p in enumerate(self.parameter_names)
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
def distance(self, a, b):
|
|
157
|
+
return sum(abs(x - y) for x, y in zip(a, b))
|
|
158
|
+
|
|
159
|
+
def get_neighbors(self, state):
|
|
160
|
+
neighbors = []
|
|
161
|
+
for i in range(len(state)):
|
|
162
|
+
for step in (-1, 1):
|
|
163
|
+
new_state = list(state)
|
|
164
|
+
new_state[i] += step
|
|
165
|
+
if 0 <= new_state[i] < len(self.values[i]):
|
|
166
|
+
new_state = tuple(new_state)
|
|
167
|
+
if new_state not in self.evaluated:
|
|
168
|
+
neighbors.append(new_state)
|
|
169
|
+
return neighbors
|
|
170
|
+
|
|
171
|
+
# ========================================================
|
|
172
|
+
# EVALUATE -- sequential, fold by fold, with pruning
|
|
173
|
+
# ========================================================
|
|
174
|
+
|
|
175
|
+
def evaluate(self, state, X, y):
|
|
176
|
+
params = self.state_to_params(state)
|
|
177
|
+
|
|
178
|
+
base_model = clone(self.estimator)
|
|
179
|
+
base_model.set_params(**params)
|
|
180
|
+
if "n_jobs" in base_model.get_params():
|
|
181
|
+
base_model.set_params(n_jobs=1)
|
|
182
|
+
|
|
183
|
+
K = self._cv_splitter.get_n_splits(X, y)
|
|
184
|
+
|
|
185
|
+
incumbent = max(self.evaluated.values()) if self.evaluated else None
|
|
186
|
+
|
|
187
|
+
fold_scores = []
|
|
188
|
+
pruned = False
|
|
189
|
+
prune_reason = None
|
|
190
|
+
|
|
191
|
+
start = time.perf_counter()
|
|
192
|
+
|
|
193
|
+
for k, (train_idx, test_idx) in enumerate(
|
|
194
|
+
self._cv_splitter.split(X, y), start=1
|
|
195
|
+
):
|
|
196
|
+
X_train, X_test = safe_index(X, train_idx), safe_index(X, test_idx)
|
|
197
|
+
y_train, y_test = safe_index(y, train_idx), safe_index(y, test_idx)
|
|
198
|
+
|
|
199
|
+
fold_model = clone(base_model)
|
|
200
|
+
fold_model.fit(X_train, y_train)
|
|
201
|
+
fold_score = float(self._scorer(fold_model, X_test, y_test))
|
|
202
|
+
fold_scores.append(fold_score)
|
|
203
|
+
|
|
204
|
+
if fold_score > self._observed_max_fold_score:
|
|
205
|
+
self._observed_max_fold_score = fold_score
|
|
206
|
+
|
|
207
|
+
partial_mean = float(np.mean(fold_scores))
|
|
208
|
+
self._partial_mean_history.setdefault(k, []).append(partial_mean)
|
|
209
|
+
|
|
210
|
+
is_last_fold = (k == K)
|
|
211
|
+
can_prune = (
|
|
212
|
+
self.enable_pruning
|
|
213
|
+
and incumbent is not None
|
|
214
|
+
and not is_last_fold
|
|
215
|
+
and k >= self.min_folds_before_pruning
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
if can_prune and self.pruning_strategy == "optimistic":
|
|
219
|
+
upper_bound = (
|
|
220
|
+
self.score_upper_bound
|
|
221
|
+
if self.score_upper_bound is not None
|
|
222
|
+
else self._observed_max_fold_score
|
|
223
|
+
)
|
|
224
|
+
best_possible_mean = (
|
|
225
|
+
sum(fold_scores) + (K - k) * upper_bound
|
|
226
|
+
) / K
|
|
227
|
+
if best_possible_mean < incumbent - self.pruning_margin:
|
|
228
|
+
pruned = True
|
|
229
|
+
prune_reason = "optimistic_bound"
|
|
230
|
+
break
|
|
231
|
+
|
|
232
|
+
elif can_prune and self.pruning_strategy == "percentile":
|
|
233
|
+
hist = self._partial_mean_history.get(k, [])
|
|
234
|
+
if len(hist) >= self.min_history_for_percentile:
|
|
235
|
+
threshold = np.percentile(hist, self.pruning_percentile)
|
|
236
|
+
if partial_mean < threshold:
|
|
237
|
+
pruned = True
|
|
238
|
+
prune_reason = "percentile_bound"
|
|
239
|
+
break
|
|
240
|
+
|
|
241
|
+
runtime = time.perf_counter() - start
|
|
242
|
+
score = float(np.mean(fold_scores))
|
|
243
|
+
|
|
244
|
+
self.evaluated[state] = score
|
|
245
|
+
self.history.append({
|
|
246
|
+
"state": state,
|
|
247
|
+
"params": params,
|
|
248
|
+
"score": score,
|
|
249
|
+
"runtime": runtime,
|
|
250
|
+
"n_folds_used": len(fold_scores),
|
|
251
|
+
"n_folds_total": K,
|
|
252
|
+
"pruned": pruned,
|
|
253
|
+
"prune_reason": prune_reason,
|
|
254
|
+
})
|
|
255
|
+
|
|
256
|
+
return score
|
|
257
|
+
|
|
258
|
+
# ========================================================
|
|
259
|
+
# SURROGATE
|
|
260
|
+
# ========================================================
|
|
261
|
+
|
|
262
|
+
def fit_surrogate(self):
|
|
263
|
+
X_train = np.array(list(self.evaluated.keys()), dtype=float)
|
|
264
|
+
y_train = np.array(list(self.evaluated.values()), dtype=float)
|
|
265
|
+
self.surrogate.fit(X_train, y_train)
|
|
266
|
+
|
|
267
|
+
def predict_uncertainty(self, states):
|
|
268
|
+
states_array = np.array(states, dtype=float)
|
|
269
|
+
mean, std = self.surrogate.predict(states_array, return_std=True)
|
|
270
|
+
return mean, std
|
|
271
|
+
|
|
272
|
+
def calculate_f(self, state, start_state):
|
|
273
|
+
"""Diagnostic blended distance/quality score. Not in the hot path."""
|
|
274
|
+
mean, std = self.predict_uncertainty([state])
|
|
275
|
+
mean = mean[0]
|
|
276
|
+
std = std[0]
|
|
277
|
+
|
|
278
|
+
upper_score = mean + (self.exploration_weight * std)
|
|
279
|
+
|
|
280
|
+
observed = np.array(list(self.evaluated.values()))
|
|
281
|
+
lo, hi = observed.min(), observed.max()
|
|
282
|
+
span = hi - lo
|
|
283
|
+
norm_score = np.clip((upper_score - lo) / span, 0.0, 1.0) if span > 0 else 0.5
|
|
284
|
+
|
|
285
|
+
g_raw = self.distance(start_state, state)
|
|
286
|
+
g = np.clip(g_raw / max(self.max_neighbor_radius, 1), 0.0, 1.0)
|
|
287
|
+
|
|
288
|
+
h = 1 - norm_score
|
|
289
|
+
return g + h
|
|
290
|
+
|
|
291
|
+
def greedy_select(self):
|
|
292
|
+
best_state = max(self.evaluated, key=self.evaluated.get)
|
|
293
|
+
neighbors = self.get_neighbors(best_state)
|
|
294
|
+
if not neighbors:
|
|
295
|
+
return None
|
|
296
|
+
mean, std = self.predict_uncertainty(neighbors)
|
|
297
|
+
upper_score = mean + (self.exploration_weight * std)
|
|
298
|
+
return neighbors[int(np.argmax(upper_score))]
|
|
299
|
+
|
|
300
|
+
def _resolve_plateau(self, start_state):
|
|
301
|
+
visited = {start_state}
|
|
302
|
+
frontier = [start_state]
|
|
303
|
+
candidates = []
|
|
304
|
+
|
|
305
|
+
for _ in range(self.max_neighbor_radius):
|
|
306
|
+
next_frontier = []
|
|
307
|
+
for state in frontier:
|
|
308
|
+
for i in range(len(state)):
|
|
309
|
+
for step in (-1, 1):
|
|
310
|
+
new_state = list(state)
|
|
311
|
+
new_state[i] += step
|
|
312
|
+
if 0 <= new_state[i] < len(self.values[i]):
|
|
313
|
+
new_state = tuple(new_state)
|
|
314
|
+
if new_state in visited:
|
|
315
|
+
continue
|
|
316
|
+
visited.add(new_state)
|
|
317
|
+
next_frontier.append(new_state)
|
|
318
|
+
if new_state not in self.evaluated:
|
|
319
|
+
candidates.append(new_state)
|
|
320
|
+
frontier = next_frontier
|
|
321
|
+
if candidates or not frontier:
|
|
322
|
+
break
|
|
323
|
+
|
|
324
|
+
if not candidates:
|
|
325
|
+
return None
|
|
326
|
+
|
|
327
|
+
mean, std = self.predict_uncertainty(candidates)
|
|
328
|
+
upper = mean + self.exploration_weight * std
|
|
329
|
+
return candidates[int(np.argmax(upper))]
|
|
330
|
+
|
|
331
|
+
# ========================================================
|
|
332
|
+
# MAIN FIT
|
|
333
|
+
# ========================================================
|
|
334
|
+
|
|
335
|
+
def fit(self, X, y):
|
|
336
|
+
total_start = time.perf_counter()
|
|
337
|
+
|
|
338
|
+
initial_size = min(
|
|
339
|
+
self.initial_points, len(self.all_states), self.max_evaluations
|
|
340
|
+
)
|
|
341
|
+
initial_states = self.rng.choice(
|
|
342
|
+
len(self.all_states), size=initial_size, replace=False
|
|
343
|
+
)
|
|
344
|
+
for index in initial_states:
|
|
345
|
+
self.evaluate(self.all_states[index], X, y)
|
|
346
|
+
|
|
347
|
+
best_so_far = max(self.evaluated.values())
|
|
348
|
+
evaluations_since_improvement = 0
|
|
349
|
+
self.stopped_early = False
|
|
350
|
+
|
|
351
|
+
while len(self.evaluated) < min(self.max_evaluations, len(self.all_states)):
|
|
352
|
+
|
|
353
|
+
if (
|
|
354
|
+
self.early_stopping_patience is not None
|
|
355
|
+
and evaluations_since_improvement >= self.early_stopping_patience
|
|
356
|
+
):
|
|
357
|
+
self.stopped_early = True
|
|
358
|
+
break
|
|
359
|
+
|
|
360
|
+
if len(self.evaluated) >= 3:
|
|
361
|
+
self.fit_surrogate()
|
|
362
|
+
|
|
363
|
+
next_state = self.greedy_select()
|
|
364
|
+
|
|
365
|
+
if next_state is None:
|
|
366
|
+
current_best = max(self.evaluated, key=self.evaluated.get)
|
|
367
|
+
next_state = self._resolve_plateau(current_best)
|
|
368
|
+
|
|
369
|
+
if next_state is None:
|
|
370
|
+
remaining = [s for s in self.all_states if s not in self.evaluated]
|
|
371
|
+
if not remaining:
|
|
372
|
+
break
|
|
373
|
+
mean, std = self.predict_uncertainty(remaining)
|
|
374
|
+
upper = mean + self.exploration_weight * std
|
|
375
|
+
next_state = remaining[int(np.argmax(upper))]
|
|
376
|
+
|
|
377
|
+
new_score = self.evaluate(next_state, X, y)
|
|
378
|
+
|
|
379
|
+
if new_score > best_so_far:
|
|
380
|
+
best_so_far = new_score
|
|
381
|
+
evaluations_since_improvement = 0
|
|
382
|
+
else:
|
|
383
|
+
evaluations_since_improvement += 1
|
|
384
|
+
|
|
385
|
+
else:
|
|
386
|
+
remaining = [s for s in self.all_states if s not in self.evaluated]
|
|
387
|
+
if not remaining:
|
|
388
|
+
break
|
|
389
|
+
random_state = remaining[self.rng.integers(len(remaining))]
|
|
390
|
+
self.evaluate(random_state, X, y)
|
|
391
|
+
|
|
392
|
+
best_state = max(self.evaluated, key=self.evaluated.get)
|
|
393
|
+
self.best_state = best_state
|
|
394
|
+
self.best_score = self.evaluated[best_state]
|
|
395
|
+
self.best_params = self.state_to_params(best_state)
|
|
396
|
+
self.n_evaluations = len(self.evaluated)
|
|
397
|
+
self.total_time = time.perf_counter() - total_start
|
|
398
|
+
|
|
399
|
+
self.n_pruned = sum(1 for h in self.history if h["pruned"])
|
|
400
|
+
self.total_folds_run = sum(h["n_folds_used"] for h in self.history)
|
|
401
|
+
self.total_folds_possible = sum(h["n_folds_total"] for h in self.history)
|
|
402
|
+
self.folds_saved = self.total_folds_possible - self.total_folds_run
|
|
403
|
+
|
|
404
|
+
return self
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Surrogate models used by AdaptiveGreedySearch to score unevaluated
|
|
3
|
+
hyperparameter configurations.
|
|
4
|
+
|
|
5
|
+
Both surrogates expose the same interface:
|
|
6
|
+
.fit(X, y)
|
|
7
|
+
.predict(states, return_std=True) -> (mean, std)
|
|
8
|
+
so AdaptiveGreedySearch never needs to know which one it's using.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
|
|
13
|
+
from sklearn.gaussian_process import GaussianProcessRegressor
|
|
14
|
+
from sklearn.gaussian_process.kernels import ConstantKernel, Matern, WhiteKernel
|
|
15
|
+
from sklearn.neighbors import KernelDensity
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TPESurrogate:
|
|
19
|
+
"""
|
|
20
|
+
Minimal Tree-structured Parzen Estimator surrogate (same idea as
|
|
21
|
+
Optuna's default sampler). Splits observed points into "good" (top
|
|
22
|
+
`gamma` fraction by score) and "bad" (the rest), fits a KDE over
|
|
23
|
+
each group in grid-coordinate space, and scores candidates by the
|
|
24
|
+
good/bad log-density ratio.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def __init__(self, gamma=0.25, bandwidth=1.0):
|
|
28
|
+
self.gamma = gamma
|
|
29
|
+
self.bandwidth = bandwidth
|
|
30
|
+
self.good_kde = None
|
|
31
|
+
self.bad_kde = None
|
|
32
|
+
self._y_min = 0.0
|
|
33
|
+
self._y_max = 1.0
|
|
34
|
+
|
|
35
|
+
def fit(self, X, y):
|
|
36
|
+
X = np.asarray(X, dtype=float)
|
|
37
|
+
y = np.asarray(y, dtype=float)
|
|
38
|
+
|
|
39
|
+
order = np.argsort(y)[::-1]
|
|
40
|
+
n_good = max(1, int(np.ceil(self.gamma * len(y))))
|
|
41
|
+
good_idx = order[:n_good]
|
|
42
|
+
bad_idx = order[n_good:]
|
|
43
|
+
if len(bad_idx) == 0:
|
|
44
|
+
bad_idx = good_idx
|
|
45
|
+
|
|
46
|
+
self._y_min, self._y_max = float(y.min()), float(y.max())
|
|
47
|
+
|
|
48
|
+
self.good_kde = KernelDensity(bandwidth=self.bandwidth).fit(X[good_idx])
|
|
49
|
+
self.bad_kde = KernelDensity(bandwidth=self.bandwidth).fit(X[bad_idx])
|
|
50
|
+
return self
|
|
51
|
+
|
|
52
|
+
def predict(self, states, return_std=True):
|
|
53
|
+
states = np.asarray(states, dtype=float)
|
|
54
|
+
|
|
55
|
+
log_l = self.good_kde.score_samples(states)
|
|
56
|
+
log_g = self.bad_kde.score_samples(states)
|
|
57
|
+
tpe_score = log_l - log_g
|
|
58
|
+
|
|
59
|
+
squashed = 1.0 / (1.0 + np.exp(-tpe_score))
|
|
60
|
+
mean = self._y_min + squashed * (self._y_max - self._y_min)
|
|
61
|
+
|
|
62
|
+
if not return_std:
|
|
63
|
+
return mean
|
|
64
|
+
|
|
65
|
+
std = (1.0 / (1.0 + np.abs(tpe_score))) * (self._y_max - self._y_min)
|
|
66
|
+
return mean, std
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def build_gp_surrogate(random_state=None):
|
|
70
|
+
"""Construct the Matern-kernel GaussianProcessRegressor surrogate."""
|
|
71
|
+
kernel = (
|
|
72
|
+
ConstantKernel(1.0, (1e-3, 1e3))
|
|
73
|
+
* Matern(length_scale=1.0, nu=2.5)
|
|
74
|
+
+ WhiteKernel(noise_level=1e-5, noise_level_bounds=(1e-8, 1e-1))
|
|
75
|
+
)
|
|
76
|
+
return GaussianProcessRegressor(
|
|
77
|
+
kernel=kernel,
|
|
78
|
+
alpha=1e-6,
|
|
79
|
+
normalize_y=True,
|
|
80
|
+
n_restarts_optimizer=3,
|
|
81
|
+
random_state=random_state,
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def make_surrogate(surrogate_type, tpe_gamma=0.25, tpe_bandwidth=1.0, random_state=None):
|
|
86
|
+
"""Factory: build the requested surrogate ("tpe" or "gp")."""
|
|
87
|
+
if surrogate_type == "tpe":
|
|
88
|
+
return TPESurrogate(gamma=tpe_gamma, bandwidth=tpe_bandwidth)
|
|
89
|
+
elif surrogate_type == "gp":
|
|
90
|
+
return build_gp_surrogate(random_state=random_state)
|
|
91
|
+
raise ValueError(f"Unknown surrogate_type '{surrogate_type}', expected 'tpe' or 'gp'")
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "adaptive-greedy-search"
|
|
7
|
+
version = "2.0.0"
|
|
8
|
+
description = "AGS (Adaptive Greedy Search): surrogate-guided hyperparameter search with plateau escape and pruned cross-validation."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = {text = "MIT"}
|
|
11
|
+
authors = [
|
|
12
|
+
{name = "Mohammad Jawad Hasan"}
|
|
13
|
+
]
|
|
14
|
+
requires-python = ">=3.8"
|
|
15
|
+
keywords = ["hyperparameter-optimization", "machine-learning", "scikit-learn", "AutoML", "TPE"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Intended Audience :: Science/Research",
|
|
21
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"numpy>=1.20",
|
|
25
|
+
"scikit-learn>=1.0",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Homepage = "https://github.com/MJwd-Hasan/adaptive-greedy-search"
|
|
30
|
+
Repository = "https://github.com/MJwd-Hasan/adaptive-greedy-search"
|
|
31
|
+
|
|
32
|
+
[tool.setuptools]
|
|
33
|
+
packages = ["ags"]
|