zeroth-learn 0.2.4__tar.gz → 0.2.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- zeroth_learn-0.2.6/PKG-INFO +92 -0
- zeroth_learn-0.2.6/README.md +78 -0
- zeroth_learn-0.2.6/pyproject.toml +38 -0
- zeroth_learn-0.2.6/tests/test_gradient_estimator.py +23 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/__init__.py +1 -1
- zeroth_learn-0.2.6/zeroth/abstract/activation.py +20 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/blackbox.py +3 -2
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/metric.py +1 -1
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/model.py +16 -13
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/neural_network.py +2 -2
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/perturbation_matrix.py +2 -2
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/summary.py +12 -7
- zeroth_learn-0.2.6/zeroth/all.py +5 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/data.py +2 -2
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/experiment.py +14 -13
- zeroth_learn-0.2.6/zeroth/first_order/__init__.py +4 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/model.py +5 -6
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/neural_network.py +6 -4
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/optimizers.py +3 -3
- zeroth_learn-0.2.6/zeroth/losses/binary_cross_entropy.py +38 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/plot_losses.py +9 -23
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/activation_functions.py +10 -11
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/dataclasses_utils.py +5 -6
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/metrics.py +1 -1
- zeroth_learn-0.2.6/zeroth/zeroth_order/__init__.py +12 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/gradient_estimators.py +2 -2
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/model.py +5 -6
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/neural_network/neural_network.py +7 -5
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/neural_network/parameter_manager.py +3 -5
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/optimizers.py +7 -7
- zeroth_learn-0.2.6/zeroth_learn.egg-info/PKG-INFO +92 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/SOURCES.txt +2 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/requires.txt +3 -0
- zeroth_learn-0.2.4/PKG-INFO +0 -7
- zeroth_learn-0.2.4/README.md +0 -274
- zeroth_learn-0.2.4/pyproject.toml +0 -18
- zeroth_learn-0.2.4/zeroth/abstract/activation.py +0 -20
- zeroth_learn-0.2.4/zeroth/all.py +0 -12
- zeroth_learn-0.2.4/zeroth/first_order/__init__.py +0 -5
- zeroth_learn-0.2.4/zeroth/zeroth_order/__init__.py +0 -6
- zeroth_learn-0.2.4/zeroth_learn.egg-info/PKG-INFO +0 -7
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/MANIFEST.in +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/setup.cfg +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/__init__.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/data_creator.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/loss.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/optimizer.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/layer.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/losses/__init__.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/losses/cross_entropy.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/losses/mse.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/paths.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/types.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/__init__.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/perturbation_matrices.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/neural_network/__init__.py +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/zeroth_order_blackbox.py +1 -1
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/dependency_links.txt +0 -0
- {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: zeroth-learn
|
|
3
|
+
Version: 0.2.6
|
|
4
|
+
Summary: Zeroth-order optimization for black-box models
|
|
5
|
+
Project-URL: Repository, https://github.com/nicolasmalet/Zeroth-Learn
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
Requires-Dist: numpy
|
|
9
|
+
Requires-Dist: pandas
|
|
10
|
+
Requires-Dist: matplotlib
|
|
11
|
+
Requires-Dist: cycler
|
|
12
|
+
Provides-Extra: mnist
|
|
13
|
+
Requires-Dist: scikit-learn; extra == "mnist"
|
|
14
|
+
|
|
15
|
+
# Zeroth-Learn
|
|
16
|
+
|
|
17
|
+
**A NumPy library for estimating gradients from loss evaluations and training black-box models.**
|
|
18
|
+
|
|
19
|
+
The library implements coordinate finite differences, random-direction estimation, and SGD/Adam updates. Its MNIST experiments check that these components can train models when backpropagation is unavailable. The estimator and optimizer interfaces can also be used with other black boxes, including the simulator in [Quantum-Learn](https://github.com/nicolasmalet/Quantum-Learn).
|
|
20
|
+
|
|
21
|
+
## 1. Optimisation without derivatives
|
|
22
|
+
|
|
23
|
+
Let $\theta\in\mathbb R^d$ be model parameters and $f(\theta)$ a mini-batch loss. The model exposes values of $f$, but not $\nabla f$. Zeroth-order optimisation estimates a descent direction by evaluating nearby parameters:
|
|
24
|
+
|
|
25
|
+
```math
|
|
26
|
+
\theta_{k+1}=\theta_k-\eta_k\widehat{\nabla f}(\theta_k).
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
For Adam, the estimated gradient enters the usual first- and second-moment update in place of an exact gradient. No derivative of the black-box model is required.
|
|
30
|
+
|
|
31
|
+
## 2. Finite-difference estimators
|
|
32
|
+
|
|
33
|
+
A one-sided coordinate difference estimates component $i$ as
|
|
34
|
+
|
|
35
|
+
```math
|
|
36
|
+
\widehat{\partial_i f}(\theta)
|
|
37
|
+
=\frac{f(\theta+\delta e_i)-f(\theta)}{\delta}.
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Estimating all $d$ components needs $d+1$ loss evaluations. [`GlobalFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) does this for every coordinate; [`PartialFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) evaluates only selected coordinates and returns zero elsewhere.
|
|
41
|
+
|
|
42
|
+
For a cheaper estimate in high dimension, [`SimultaneousPerturbation`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) draws $T$ Rademacher directions. At construction, the matrix $P\in\mathbb R^{T\times d}$ has entries $P_{ij}=s_{ij}/\sqrt T$, where each $s_{ij}$ is $+1$ or $-1$ with equal probability. The implementation uses
|
|
43
|
+
|
|
44
|
+
```math
|
|
45
|
+
\widehat{\nabla f}(\theta)
|
|
46
|
+
=\frac{1}{\delta}P^\top
|
|
47
|
+
\begin{pmatrix}
|
|
48
|
+
f(\theta+\delta P_1)-f(\theta)\\
|
|
49
|
+
\vdots\\
|
|
50
|
+
f(\theta+\delta P_T)-f(\theta)
|
|
51
|
+
\end{pmatrix}.
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Because $\mathbb E[P^\top P]=I$, the first-order term recovers $\nabla f$ in expectation. Each update costs $T+1$ loss evaluations, independent of $d$ for a fixed $T$. Larger $T$ averages more directions but increases evaluation cost and memory.
|
|
55
|
+
|
|
56
|
+
## 3. From an estimator to a model update
|
|
57
|
+
|
|
58
|
+
The interfaces separate three operations:
|
|
59
|
+
|
|
60
|
+
| Operation | Implementation |
|
|
61
|
+
| ---------------------------------------------------------- | --------------------------------------------------------------------------- |
|
|
62
|
+
| Generate perturbed parameters and reconstruct the gradient | [`gradient_estimators.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) |
|
|
63
|
+
| Generate Rademacher directions | [`perturbation_matrices.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/utils/perturbation_matrices.py) |
|
|
64
|
+
| Evaluate a model at nominal and perturbed parameters | [`zeroth_order_blackbox.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/zeroth_order_blackbox.py) |
|
|
65
|
+
| Apply SGD or Adam to the estimated gradient | [`optimizers.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/optimizers.py) |
|
|
66
|
+
|
|
67
|
+
The included [`ZerothOrderNeuralNetwork`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/neural_network.py) is one black-box implementation. [`ParameterManager`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/parameter_manager.py) maps weights and biases to a flat $\theta$; NumPy broadcasting evaluates its $T+1$ perturbed networks in one batch. This batching reduces Python overhead, not the number of loss evaluations. Other models choose how to perform those evaluations.
|
|
68
|
+
|
|
69
|
+
## 4. Numerical checks
|
|
70
|
+
|
|
71
|
+
A deterministic [test](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/tests/test_gradient_estimator.py) applies the random-direction estimator to a linear function, whose gradient is known. The MNIST experiments provide end-to-end checks of zeroth-order training on CPU. Their protocols, results, plots, and raw loss traces are in [Experiments](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/EXPERIMENTS.md).
|
|
72
|
+
|
|
73
|
+
## 5. Reproduce the experiments
|
|
74
|
+
|
|
75
|
+
With [uv](https://docs.astral.sh/uv/), the lockfile fixes the dependency versions. The first experiment run downloads MNIST from OpenML.
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
git clone https://github.com/nicolasmalet/Zeroth-Learn.git
|
|
79
|
+
cd Zeroth-Learn
|
|
80
|
+
uv sync --locked --extra mnist
|
|
81
|
+
uv run --locked --extra mnist python -m unittest discover -s tests
|
|
82
|
+
MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist
|
|
83
|
+
MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist.run_experiment nb_perturbations_vs_model_size
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
The experiments write accuracy, loss traces, and plots under `results/`.
|
|
87
|
+
|
|
88
|
+
Without uv, install the package with its MNIST extra in a virtual environment, then run the same Python modules.
|
|
89
|
+
|
|
90
|
+
## 6. Scope
|
|
91
|
+
|
|
92
|
+
The MNIST experiments validate classical models. They are not benchmarks of quantum hardware or of PyTorch, and their saved plots do not contain repeated trials or uncertainty estimates. Memory use grows with the number of perturbations in the vectorized neural-network implementation.
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
# Zeroth-Learn
|
|
2
|
+
|
|
3
|
+
**A NumPy library for estimating gradients from loss evaluations and training black-box models.**
|
|
4
|
+
|
|
5
|
+
The library implements coordinate finite differences, random-direction estimation, and SGD/Adam updates. Its MNIST experiments check that these components can train models when backpropagation is unavailable. The estimator and optimizer interfaces can also be used with other black boxes, including the simulator in [Quantum-Learn](https://github.com/nicolasmalet/Quantum-Learn).
|
|
6
|
+
|
|
7
|
+
## 1. Optimisation without derivatives
|
|
8
|
+
|
|
9
|
+
Let $\theta\in\mathbb R^d$ be model parameters and $f(\theta)$ a mini-batch loss. The model exposes values of $f$, but not $\nabla f$. Zeroth-order optimisation estimates a descent direction by evaluating nearby parameters:
|
|
10
|
+
|
|
11
|
+
```math
|
|
12
|
+
\theta_{k+1}=\theta_k-\eta_k\widehat{\nabla f}(\theta_k).
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
For Adam, the estimated gradient enters the usual first- and second-moment update in place of an exact gradient. No derivative of the black-box model is required.
|
|
16
|
+
|
|
17
|
+
## 2. Finite-difference estimators
|
|
18
|
+
|
|
19
|
+
A one-sided coordinate difference estimates component $i$ as
|
|
20
|
+
|
|
21
|
+
```math
|
|
22
|
+
\widehat{\partial_i f}(\theta)
|
|
23
|
+
=\frac{f(\theta+\delta e_i)-f(\theta)}{\delta}.
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Estimating all $d$ components needs $d+1$ loss evaluations. [`GlobalFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) does this for every coordinate; [`PartialFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) evaluates only selected coordinates and returns zero elsewhere.
|
|
27
|
+
|
|
28
|
+
For a cheaper estimate in high dimension, [`SimultaneousPerturbation`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) draws $T$ Rademacher directions. At construction, the matrix $P\in\mathbb R^{T\times d}$ has entries $P_{ij}=s_{ij}/\sqrt T$, where each $s_{ij}$ is $+1$ or $-1$ with equal probability. The implementation uses
|
|
29
|
+
|
|
30
|
+
```math
|
|
31
|
+
\widehat{\nabla f}(\theta)
|
|
32
|
+
=\frac{1}{\delta}P^\top
|
|
33
|
+
\begin{pmatrix}
|
|
34
|
+
f(\theta+\delta P_1)-f(\theta)\\
|
|
35
|
+
\vdots\\
|
|
36
|
+
f(\theta+\delta P_T)-f(\theta)
|
|
37
|
+
\end{pmatrix}.
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Because $\mathbb E[P^\top P]=I$, the first-order term recovers $\nabla f$ in expectation. Each update costs $T+1$ loss evaluations, independent of $d$ for a fixed $T$. Larger $T$ averages more directions but increases evaluation cost and memory.
|
|
41
|
+
|
|
42
|
+
## 3. From an estimator to a model update
|
|
43
|
+
|
|
44
|
+
The interfaces separate three operations:
|
|
45
|
+
|
|
46
|
+
| Operation | Implementation |
|
|
47
|
+
| ---------------------------------------------------------- | --------------------------------------------------------------------------- |
|
|
48
|
+
| Generate perturbed parameters and reconstruct the gradient | [`gradient_estimators.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) |
|
|
49
|
+
| Generate Rademacher directions | [`perturbation_matrices.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/utils/perturbation_matrices.py) |
|
|
50
|
+
| Evaluate a model at nominal and perturbed parameters | [`zeroth_order_blackbox.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/zeroth_order_blackbox.py) |
|
|
51
|
+
| Apply SGD or Adam to the estimated gradient | [`optimizers.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/optimizers.py) |
|
|
52
|
+
|
|
53
|
+
The included [`ZerothOrderNeuralNetwork`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/neural_network.py) is one black-box implementation. [`ParameterManager`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/parameter_manager.py) maps weights and biases to a flat $\theta$; NumPy broadcasting evaluates its $T+1$ perturbed networks in one batch. This batching reduces Python overhead, not the number of loss evaluations. Other models choose how to perform those evaluations.
|
|
54
|
+
|
|
55
|
+
## 4. Numerical checks
|
|
56
|
+
|
|
57
|
+
A deterministic [test](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/tests/test_gradient_estimator.py) applies the random-direction estimator to a linear function, whose gradient is known. The MNIST experiments provide end-to-end checks of zeroth-order training on CPU. Their protocols, results, plots, and raw loss traces are in [Experiments](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/EXPERIMENTS.md).
|
|
58
|
+
|
|
59
|
+
## 5. Reproduce the experiments
|
|
60
|
+
|
|
61
|
+
With [uv](https://docs.astral.sh/uv/), the lockfile fixes the dependency versions. The first experiment run downloads MNIST from OpenML.
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
git clone https://github.com/nicolasmalet/Zeroth-Learn.git
|
|
65
|
+
cd Zeroth-Learn
|
|
66
|
+
uv sync --locked --extra mnist
|
|
67
|
+
uv run --locked --extra mnist python -m unittest discover -s tests
|
|
68
|
+
MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist
|
|
69
|
+
MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist.run_experiment nb_perturbations_vs_model_size
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The experiments write accuracy, loss traces, and plots under `results/`.
|
|
73
|
+
|
|
74
|
+
Without uv, install the package with its MNIST extra in a virtual environment, then run the same Python modules.
|
|
75
|
+
|
|
76
|
+
## 6. Scope
|
|
77
|
+
|
|
78
|
+
The MNIST experiments validate classical models. They are not benchmarks of quantum hardware or of PyTorch, and their saved plots do not contain repeated trials or uncertainty estimates. Memory use grows with the number of perturbations in the vectorized neural-network implementation.
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "zeroth-learn"
|
|
7
|
+
version = "0.2.6"
|
|
8
|
+
description = "Zeroth-order optimization for black-box models"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"numpy",
|
|
13
|
+
"pandas",
|
|
14
|
+
"matplotlib",
|
|
15
|
+
"cycler",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.optional-dependencies]
|
|
19
|
+
mnist = ["scikit-learn"]
|
|
20
|
+
|
|
21
|
+
[project.urls]
|
|
22
|
+
Repository = "https://github.com/nicolasmalet/Zeroth-Learn"
|
|
23
|
+
|
|
24
|
+
[tool.setuptools.packages.find]
|
|
25
|
+
where = ["."]
|
|
26
|
+
include = ["zeroth*"]
|
|
27
|
+
exclude = ["lab*"]
|
|
28
|
+
|
|
29
|
+
[tool.ruff]
|
|
30
|
+
line-length = 120
|
|
31
|
+
|
|
32
|
+
[tool.ruff.lint]
|
|
33
|
+
select = ["E4", "E7", "E9", "F", "I", "UP", "B"]
|
|
34
|
+
|
|
35
|
+
[tool.ruff.lint.per-file-ignores]
|
|
36
|
+
"**/__init__.py" = ["F401", "F403"]
|
|
37
|
+
"zeroth/all.py" = ["F403"]
|
|
38
|
+
"lab/mnist/neural_networks.py" = ["E741"]
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import unittest
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
from zeroth.utils.perturbation_matrices import RademacherMatrix
|
|
6
|
+
from zeroth.zeroth_order.gradient_estimators import SimultaneousPerturbation
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class SimultaneousPerturbationTest(unittest.TestCase):
|
|
10
|
+
def test_recovers_linear_gradient(self):
|
|
11
|
+
np.random.seed(42)
|
|
12
|
+
expected = np.array([1.5, -2.0, 0.5])
|
|
13
|
+
theta = np.array([0.2, -0.1, 0.3])
|
|
14
|
+
estimator = SimultaneousPerturbation(1e-6, 2_000, RademacherMatrix(), theta.size)
|
|
15
|
+
|
|
16
|
+
perturbed_theta = estimator.perturb(theta)
|
|
17
|
+
losses = (perturbed_theta @ expected)[:, None]
|
|
18
|
+
|
|
19
|
+
np.testing.assert_allclose(estimator.get_gradient(losses), expected, atol=0.06)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
if __name__ == "__main__":
|
|
23
|
+
unittest.main()
|
|
@@ -4,7 +4,7 @@ from .data_creator import DataCreator
|
|
|
4
4
|
from .loss import Loss
|
|
5
5
|
from .metric import Metric
|
|
6
6
|
from .model import Model, ModelConfig
|
|
7
|
-
from .neural_network import
|
|
7
|
+
from .neural_network import NeuralNetwork, NeuralNetworkConfig
|
|
8
8
|
from .optimizer import Optimizer
|
|
9
9
|
from .perturbation_matrix import PerturbationMatrix
|
|
10
10
|
from .summary import Summary
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
|
|
3
|
+
from ..types import Array
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Activation(ABC):
|
|
7
|
+
"""Base class for activation functions."""
|
|
8
|
+
|
|
9
|
+
def __repr__(self) -> str:
|
|
10
|
+
return f"{self.__class__.__name__}()"
|
|
11
|
+
|
|
12
|
+
@abstractmethod
|
|
13
|
+
def __call__(self, x: Array) -> Array:
|
|
14
|
+
"""Apply the activation function."""
|
|
15
|
+
...
|
|
16
|
+
|
|
17
|
+
@abstractmethod
|
|
18
|
+
def derivative(self, x: Array) -> Array:
|
|
19
|
+
"""Compute the activation derivative."""
|
|
20
|
+
...
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import Any
|
|
2
3
|
|
|
3
4
|
from ..types import Array
|
|
4
5
|
|
|
@@ -24,7 +25,7 @@ class BlackBox(ABC):
|
|
|
24
25
|
...
|
|
25
26
|
|
|
26
27
|
@abstractmethod
|
|
27
|
-
def init_params(self, params: dict) -> None:
|
|
28
|
+
def init_params(self, params: dict[str, Any]) -> None:
|
|
28
29
|
"""Manually initializes the weights and biases of the network.
|
|
29
30
|
|
|
30
31
|
Args:
|
|
@@ -33,7 +34,7 @@ class BlackBox(ABC):
|
|
|
33
34
|
...
|
|
34
35
|
|
|
35
36
|
@abstractmethod
|
|
36
|
-
def get_params(self) -> dict:
|
|
37
|
+
def get_params(self) -> dict[str, Any]:
|
|
37
38
|
"""Retrieves the current parameters of the network.
|
|
38
39
|
|
|
39
40
|
Returns:
|
|
@@ -4,19 +4,21 @@ import pickle
|
|
|
4
4
|
from abc import ABC, abstractmethod
|
|
5
5
|
from dataclasses import dataclass
|
|
6
6
|
from pathlib import Path
|
|
7
|
-
from typing import
|
|
7
|
+
from typing import Any
|
|
8
8
|
|
|
9
9
|
import matplotlib.pyplot as plt
|
|
10
10
|
import numpy as np
|
|
11
11
|
import pandas as pd
|
|
12
|
+
from matplotlib.figure import Figure
|
|
12
13
|
|
|
14
|
+
from ..data import Data
|
|
15
|
+
from ..plot_losses import plot_losses
|
|
16
|
+
from ..types import Array
|
|
13
17
|
from .blackbox import BlackBox
|
|
14
18
|
from .loss import Loss
|
|
19
|
+
from .metric import Metric
|
|
15
20
|
from .optimizer import Optimizer
|
|
16
21
|
from .summary import Summary
|
|
17
|
-
from ..data import Data
|
|
18
|
-
from ..plot_losses import plot_losses
|
|
19
|
-
from ..types import Array
|
|
20
22
|
|
|
21
23
|
|
|
22
24
|
@dataclass(frozen=True, kw_only=True)
|
|
@@ -30,9 +32,9 @@ class ModelConfig(ABC, Summary):
|
|
|
30
32
|
nb_epochs (int): Number of passes through the entire dataset.
|
|
31
33
|
"""
|
|
32
34
|
name: str
|
|
33
|
-
id: dict
|
|
35
|
+
id: dict[str, Any]
|
|
34
36
|
loss: Loss
|
|
35
|
-
metric:
|
|
37
|
+
metric: Metric
|
|
36
38
|
batch_size: int
|
|
37
39
|
nb_epochs: int = 1
|
|
38
40
|
|
|
@@ -60,10 +62,10 @@ class Model(ABC):
|
|
|
60
62
|
self.config = config
|
|
61
63
|
|
|
62
64
|
self.name: str = config.name
|
|
63
|
-
self.id: dict = config.id
|
|
65
|
+
self.id: dict[str, Any] = config.id
|
|
64
66
|
self.data: Data = data
|
|
65
67
|
self.loss: Loss = config.loss
|
|
66
|
-
self.metric:
|
|
68
|
+
self.metric: Metric = config.metric
|
|
67
69
|
self.batch_size: int = config.batch_size
|
|
68
70
|
self.nb_epochs: int = config.nb_epochs
|
|
69
71
|
|
|
@@ -91,7 +93,7 @@ class Model(ABC):
|
|
|
91
93
|
print_indexes = np.linspace(0, nb_batches - 1, nb_print).astype(int)
|
|
92
94
|
|
|
93
95
|
for epoch_idx in range(self.nb_epochs):
|
|
94
|
-
print(f" epoch
|
|
96
|
+
print(f" epoch {epoch_idx + 1} of {self.nb_epochs}")
|
|
95
97
|
self.data.permutation()
|
|
96
98
|
self.data.batch_size = self.batch_size
|
|
97
99
|
|
|
@@ -100,12 +102,12 @@ class Model(ABC):
|
|
|
100
102
|
self.training_loss[epoch_idx * nb_batches + batch_idx] = avg_loss
|
|
101
103
|
|
|
102
104
|
if batch_idx in print_indexes:
|
|
103
|
-
print(f" batch
|
|
105
|
+
print(f" batch {batch_idx + 1} of {nb_batches}, "
|
|
104
106
|
f"loss : {np.round(self.training_loss[epoch_idx * nb_batches + batch_idx], 3)}")
|
|
105
107
|
|
|
106
108
|
self.test()
|
|
107
109
|
|
|
108
|
-
def plot_loss(self, smooth_fraction: float = 0.05) ->
|
|
110
|
+
def plot_loss(self, smooth_fraction: float = 0.05) -> Figure:
|
|
109
111
|
fig = plot_losses(dimension=0,
|
|
110
112
|
models=[self],
|
|
111
113
|
title=self.name,
|
|
@@ -113,7 +115,7 @@ class Model(ABC):
|
|
|
113
115
|
plt.close(fig)
|
|
114
116
|
return fig
|
|
115
117
|
|
|
116
|
-
def test(self) ->
|
|
118
|
+
def test(self) -> Array:
|
|
117
119
|
X_test, Y_true = self.data.X_test, self.data.Y_test # (in, batch), (out, batch)
|
|
118
120
|
Y_pred = self.neural_network(X_test) # (out, batch)
|
|
119
121
|
|
|
@@ -121,6 +123,7 @@ class Model(ABC):
|
|
|
121
123
|
self.test_loss = self.loss.compute_loss(Y_pred, Y_true)
|
|
122
124
|
|
|
123
125
|
print(f" {self.id} accuracy : {self.test_accuracy}, loss : {self.test_loss}")
|
|
126
|
+
return Y_pred
|
|
124
127
|
|
|
125
128
|
def save_loss(self, save_dir: Path) -> None:
|
|
126
129
|
save_dir.mkdir(parents=True, exist_ok=True)
|
|
@@ -140,7 +143,7 @@ class Model(ABC):
|
|
|
140
143
|
pickle.dump(params_dict, f)
|
|
141
144
|
|
|
142
145
|
def load_weights(self, load_dir: Path) -> None:
|
|
143
|
-
"""
|
|
146
|
+
"""Restore model parameters from a pickle file."""
|
|
144
147
|
load_path = load_dir / self.WEIGHTS_FILE
|
|
145
148
|
with load_path.open('rb') as f:
|
|
146
149
|
params_dict = pickle.load(f)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
from abc import ABC
|
|
2
1
|
from dataclasses import dataclass
|
|
3
2
|
|
|
4
3
|
from .activation import Activation
|
|
4
|
+
from .blackbox import BlackBox
|
|
5
5
|
from .summary import Summary
|
|
6
6
|
|
|
7
7
|
|
|
@@ -12,7 +12,7 @@ class NeuralNetworkConfig(Summary):
|
|
|
12
12
|
activations: list[Activation]
|
|
13
13
|
|
|
14
14
|
|
|
15
|
-
class NeuralNetwork(
|
|
15
|
+
class NeuralNetwork(BlackBox):
|
|
16
16
|
def __init__(self, config: NeuralNetworkConfig, input_dim: int, output_dim: int):
|
|
17
17
|
self.name: str = config.name
|
|
18
18
|
self.nb_layers: int = len(config.activations)
|
|
@@ -4,12 +4,12 @@ from ..types import Array
|
|
|
4
4
|
|
|
5
5
|
|
|
6
6
|
class PerturbationMatrix(ABC):
|
|
7
|
-
"""
|
|
7
|
+
"""Base class for perturbation-matrix generators."""
|
|
8
8
|
|
|
9
9
|
def __repr__(self) -> str:
|
|
10
10
|
return f"{self.__class__.__name__}()"
|
|
11
11
|
|
|
12
12
|
@abstractmethod
|
|
13
13
|
def __call__(self, nb_perturbation: int, nb_parameters: int) -> Array:
|
|
14
|
-
"""
|
|
14
|
+
"""Generate perturbation directions."""
|
|
15
15
|
...
|
|
@@ -16,17 +16,20 @@ class Summary:
|
|
|
16
16
|
print(self._summary(self, indent=0), file=f)
|
|
17
17
|
|
|
18
18
|
@classmethod
|
|
19
|
-
def load(cls, path: str, context: dict = None) -> Summary:
|
|
20
|
-
with open(path
|
|
21
|
-
code = "".join(
|
|
22
|
-
|
|
19
|
+
def load(cls, path: str, context: dict[str, Any] | None = None) -> Summary:
|
|
20
|
+
with open(path) as f:
|
|
21
|
+
code = "".join(line for line in f)
|
|
22
|
+
result = eval(code.strip(), {"__builtins__": __builtins__}, context)
|
|
23
|
+
if not isinstance(result, cls):
|
|
24
|
+
raise TypeError(f"Expected {cls.__name__}, got {type(result).__name__}")
|
|
25
|
+
return result
|
|
23
26
|
|
|
24
27
|
def _summary(self, obj: Any, indent: int) -> str:
|
|
25
28
|
shift = " " * indent
|
|
26
29
|
next_shift = " " * (indent + 1)
|
|
27
30
|
|
|
28
31
|
if dataclasses.is_dataclass(obj):
|
|
29
|
-
cls_name = obj.
|
|
32
|
+
cls_name = type(obj).__name__
|
|
30
33
|
items = []
|
|
31
34
|
for f in dataclasses.fields(obj):
|
|
32
35
|
val = getattr(obj, f.name)
|
|
@@ -37,13 +40,15 @@ class Summary:
|
|
|
37
40
|
return f"{cls_name}(\n{content}\n{shift})"
|
|
38
41
|
|
|
39
42
|
elif isinstance(obj, list):
|
|
40
|
-
if not obj:
|
|
43
|
+
if not obj:
|
|
44
|
+
return "[]"
|
|
41
45
|
items = [self._summary(item, indent + 1) for item in obj]
|
|
42
46
|
content = ",\n".join([f"{next_shift}{item.lstrip()}" for item in items])
|
|
43
47
|
return f"[\n{content}\n{shift}]"
|
|
44
48
|
|
|
45
49
|
elif isinstance(obj, dict):
|
|
46
|
-
if not obj:
|
|
50
|
+
if not obj:
|
|
51
|
+
return "{}"
|
|
47
52
|
items = [f"{next_shift}{repr(k)}: {self._summary(v, indent + 1).lstrip()}" for k, v in obj.items()]
|
|
48
53
|
content = ",\n".join(items)
|
|
49
54
|
return f"{{\n{content}\n{shift}}}"
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
from
|
|
1
|
+
from collections.abc import Iterator
|
|
2
2
|
|
|
3
3
|
import numpy as np
|
|
4
4
|
|
|
@@ -17,7 +17,7 @@ class Data:
|
|
|
17
17
|
self.input_dim: int = self.raw_X_train.shape[1]
|
|
18
18
|
self.output_dim: int = self.raw_Y_train.shape[1] if nb_class == 0 else nb_class
|
|
19
19
|
|
|
20
|
-
self.batch_size: int
|
|
20
|
+
self.batch_size: int = 1
|
|
21
21
|
|
|
22
22
|
self.indices: np.ndarray = np.arange(self.nb_data)
|
|
23
23
|
|
|
@@ -1,14 +1,15 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
3
|
import itertools
|
|
4
|
-
from pathlib import Path
|
|
5
4
|
from dataclasses import dataclass, replace
|
|
6
|
-
from
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any, cast
|
|
7
7
|
|
|
8
8
|
import matplotlib.pyplot as plt
|
|
9
9
|
import pandas as pd
|
|
10
|
+
from matplotlib.figure import Figure
|
|
10
11
|
|
|
11
|
-
from .abstract import Model, ModelConfig,
|
|
12
|
+
from .abstract import DataCreator, Model, ModelConfig, Summary
|
|
12
13
|
from .data import Data
|
|
13
14
|
from .plot_losses import plot_losses
|
|
14
15
|
from .utils.dataclasses_utils import get_name, set_value_by_path
|
|
@@ -18,7 +19,7 @@ from .utils.dataclasses_utils import get_name, set_value_by_path
|
|
|
18
19
|
class VariationConfig:
|
|
19
20
|
name: str
|
|
20
21
|
param: list[str]
|
|
21
|
-
values:
|
|
22
|
+
values: list[list[Any]]
|
|
22
23
|
|
|
23
24
|
|
|
24
25
|
@dataclass(frozen=True)
|
|
@@ -54,11 +55,11 @@ class Experiment:
|
|
|
54
55
|
self.models: list[Model] = generate_models(config.base_model, config.variations, self.data)
|
|
55
56
|
|
|
56
57
|
def train_models(self, nb_print: int) -> None:
|
|
57
|
-
print(
|
|
58
|
+
print("Training Models")
|
|
58
59
|
for model in self.models:
|
|
59
60
|
model.train(nb_print)
|
|
60
61
|
|
|
61
|
-
def plot_losses(self, title: str, plot_dimension: int, smooth_fraction: float = 0) ->
|
|
62
|
+
def plot_losses(self, title: str, plot_dimension: int, smooth_fraction: float = 0) -> Figure:
|
|
62
63
|
fig = plot_losses(title=title,
|
|
63
64
|
dimension=plot_dimension,
|
|
64
65
|
models=self.models,
|
|
@@ -69,7 +70,7 @@ class Experiment:
|
|
|
69
70
|
return fig
|
|
70
71
|
|
|
71
72
|
def test_models(self) -> None:
|
|
72
|
-
print(
|
|
73
|
+
print("Testing Models")
|
|
73
74
|
for model in self.models:
|
|
74
75
|
model.test()
|
|
75
76
|
|
|
@@ -88,7 +89,7 @@ class Experiment:
|
|
|
88
89
|
df.to_csv(save_path)
|
|
89
90
|
|
|
90
91
|
def save_weights(self, save_dir: Path) -> None:
|
|
91
|
-
for
|
|
92
|
+
for model in self.models:
|
|
92
93
|
save_path = save_dir / model.name
|
|
93
94
|
model.save_weights(save_path)
|
|
94
95
|
|
|
@@ -97,7 +98,7 @@ class Experiment:
|
|
|
97
98
|
config_path = save_dir / self.CONFIG_FILE
|
|
98
99
|
self.config.save(config_path)
|
|
99
100
|
|
|
100
|
-
for
|
|
101
|
+
for model in self.models:
|
|
101
102
|
save_path = save_dir / model.name
|
|
102
103
|
model.config.save(save_path)
|
|
103
104
|
|
|
@@ -109,13 +110,13 @@ def generate_models(base_model: ModelConfig, variations: list[VariationConfig],
|
|
|
109
110
|
|
|
110
111
|
for combination in itertools.product(*values_lists):
|
|
111
112
|
id_ = {}
|
|
112
|
-
current_model = base_model
|
|
113
|
+
current_model: ModelConfig = base_model
|
|
113
114
|
|
|
114
|
-
for var_config, current_vals in zip(variations, combination):
|
|
115
|
+
for var_config, current_vals in zip(variations, combination, strict=True):
|
|
115
116
|
id_[var_config.name] = get_name(current_vals[0])
|
|
116
117
|
|
|
117
|
-
for path, val in zip(var_config.param, current_vals):
|
|
118
|
-
current_model = set_value_by_path(current_model, path, val)
|
|
118
|
+
for path, val in zip(var_config.param, current_vals, strict=True):
|
|
119
|
+
current_model = cast(ModelConfig, set_value_by_path(current_model, path, val))
|
|
119
120
|
|
|
120
121
|
current_model = replace(current_model, id=id_)
|
|
121
122
|
models.append(current_model.instantiate(data))
|
|
@@ -2,10 +2,10 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
from dataclasses import dataclass
|
|
4
4
|
|
|
5
|
-
from
|
|
6
|
-
from .optimizers import FirstOrderOptimizerConfig, FirstOrderOptimizer
|
|
7
|
-
from ..abstract import ModelConfig, Model, NeuralNetworkConfig
|
|
5
|
+
from ..abstract import Model, ModelConfig, NeuralNetworkConfig
|
|
8
6
|
from ..data import Data
|
|
7
|
+
from .neural_network import FirstOrderNeuralNetwork
|
|
8
|
+
from .optimizers import FirstOrderOptimizerConfig
|
|
9
9
|
|
|
10
10
|
|
|
11
11
|
@dataclass(frozen=True, kw_only=True)
|
|
@@ -21,6 +21,5 @@ class FirstOrderModel(Model):
|
|
|
21
21
|
def __init__(self, config: FirstOrderModelConfig, data: Data) -> None:
|
|
22
22
|
super().__init__(config, data)
|
|
23
23
|
|
|
24
|
-
self.neural_network
|
|
25
|
-
|
|
26
|
-
self.optimizer: FirstOrderOptimizer = config.optimizer_config.instantiate()
|
|
24
|
+
self.neural_network = FirstOrderNeuralNetwork(config.neural_network_config, data.input_dim, data.output_dim)
|
|
25
|
+
self.optimizer = config.optimizer_config.instantiate()
|
|
@@ -1,6 +1,8 @@
|
|
|
1
|
-
from
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
2
3
|
from ..abstract.neural_network import NeuralNetwork, NeuralNetworkConfig
|
|
3
4
|
from ..types import Array
|
|
5
|
+
from .layer import Layer
|
|
4
6
|
|
|
5
7
|
|
|
6
8
|
class FirstOrderNeuralNetwork(NeuralNetwork):
|
|
@@ -25,14 +27,14 @@ class FirstOrderNeuralNetwork(NeuralNetwork):
|
|
|
25
27
|
def __call__(self, X: Array) -> Array:
|
|
26
28
|
return self.forward(X)
|
|
27
29
|
|
|
28
|
-
def init_params(self, params: dict) -> None:
|
|
30
|
+
def init_params(self, params: dict[str, Any]) -> None:
|
|
29
31
|
Ws = params["Ws"]
|
|
30
32
|
Bs = params["Bs"]
|
|
31
|
-
for layer, W, B in zip(self.layers, Ws, Bs):
|
|
33
|
+
for layer, W, B in zip(self.layers, Ws, Bs, strict=True):
|
|
32
34
|
layer.W = W
|
|
33
35
|
layer.B = B
|
|
34
36
|
|
|
35
|
-
def get_params(self) -> dict:
|
|
37
|
+
def get_params(self) -> dict[str, Any]:
|
|
36
38
|
Ws, Bs = [], []
|
|
37
39
|
for layer in self.layers:
|
|
38
40
|
Ws.append(layer.W)
|