zeroth-learn 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {zeroth_learn-0.2.1/zeroth_learn.egg-info → zeroth_learn-0.2.2}/PKG-INFO +7 -7
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/README.md +274 -274
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/pyproject.toml +17 -17
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/setup.cfg +4 -4
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/__init__.py +10 -10
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/activation.py +20 -20
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/blackbox.py +42 -42
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/data_creator.py +12 -12
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/loss.py +58 -58
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/metric.py +13 -13
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/model.py +147 -147
- zeroth_learn-0.2.2/zeroth/abstract/neural_network.py +25 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/optimizer.py +29 -29
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/perturbation_matrix.py +15 -15
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/abstract/summary.py +13 -13
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/data.py +32 -29
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/experiment.py +151 -151
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/first_order/__init__.py +5 -5
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/first_order/layer.py +84 -84
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/first_order/model.py +25 -24
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/first_order/neural_network.py +51 -49
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/first_order/optimizers.py +150 -150
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/losses/__init__.py +2 -2
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/losses/cross_entropy.py +41 -41
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/losses/mse.py +36 -36
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/paths.py +10 -10
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/plot_losses.py +222 -222
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/types.py +3 -3
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/utils/activation_functions.py +46 -46
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/utils/dataclasses_utils.py +35 -35
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/utils/metrics.py +11 -11
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/utils/perturbation_matrices.py +13 -13
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/__init__.py +6 -6
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/gradient_estimators.py +154 -154
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/model.py +29 -28
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/neural_network/neural_network.py +75 -72
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/neural_network/parameter_manager.py +95 -95
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/optimizers.py +126 -126
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/zeroth_order_blackbox.py +16 -16
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2/zeroth_learn.egg-info}/PKG-INFO +7 -7
- zeroth_learn-0.2.1/zeroth/abstract/neural_network.py +0 -26
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/MANIFEST.in +0 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/__init__.py +0 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/utils/__init__.py +0 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth/zeroth_order/neural_network/__init__.py +0 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth_learn.egg-info/SOURCES.txt +0 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth_learn.egg-info/dependency_links.txt +0 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth_learn.egg-info/requires.txt +0 -0
- {zeroth_learn-0.2.1 → zeroth_learn-0.2.2}/zeroth_learn.egg-info/top_level.txt +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: zeroth-learn
|
|
3
|
-
Version: 0.2.
|
|
4
|
-
Requires-Dist: numpy
|
|
5
|
-
Requires-Dist: pandas
|
|
6
|
-
Requires-Dist: matplotlib
|
|
7
|
-
Requires-Dist: cycler
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: zeroth-learn
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Requires-Dist: numpy
|
|
5
|
+
Requires-Dist: pandas
|
|
6
|
+
Requires-Dist: matplotlib
|
|
7
|
+
Requires-Dist: cycler
|
|
@@ -1,274 +1,274 @@
|
|
|
1
|
-
# Zeroth-Learn
|
|
2
|
-
|
|
3
|
-
**A research library for zeroth-order optimization (gradient-free) in machine learning, with applications to quantum
|
|
4
|
-
computing.**
|
|
5
|
-
|
|
6
|
-
*Research project — Nicolas, X24*
|
|
7
|
-
|
|
8
|
-
---
|
|
9
|
-
|
|
10
|
-
## Context & Motivation
|
|
11
|
-
|
|
12
|
-
This project originated from a fundamental question in quantum machine learning:
|
|
13
|
-
|
|
14
|
-
> **How do you train parameterized quantum circuits when backpropagation is impossible?**
|
|
15
|
-
|
|
16
|
-
Quantum circuits are too complex to differentiate we treat it as a black box, thus we need to find alternatives to
|
|
17
|
-
backpropagation, SPSA is an excellent candidate.
|
|
18
|
-
|
|
19
|
-
**SPSA (Simultaneous Perturbation Stochastic Approximation)** solves this by estimating gradients from only O(1)
|
|
20
|
-
evaluations per iteration.
|
|
21
|
-
|
|
22
|
-
Before deploying on quantum simulators, I built this library to:
|
|
23
|
-
|
|
24
|
-
1. Understand the theoretical foundations of zeroth-order optimization
|
|
25
|
-
2. Validate SPSA stability on classical benchmarks (MNIST)
|
|
26
|
-
3. Validate my results with backpropagation models.
|
|
27
|
-
|
|
28
|
-
---
|
|
29
|
-
|
|
30
|
-
## Technical Implementation
|
|
31
|
-
|
|
32
|
-
### Architecture Decisions
|
|
33
|
-
|
|
34
|
-
**Problem**: Standard deep learning frameworks (PyTorch, JAX) are tightly coupled to automatic differentiation. I needed
|
|
35
|
-
an architecture where gradient computation is a **swappable abstraction**.
|
|
36
|
-
|
|
37
|
-
**Solution**: Clean separation of concerns using abstract base classes:
|
|
38
|
-
|
|
39
|
-
```
|
|
40
|
-
Model (training loop orchestration)
|
|
41
|
-
├── NeuralNetwork (forward pass interface)
|
|
42
|
-
│ ├── NeuralNetworkBackpropagation (layer-based, stores activations)
|
|
43
|
-
│ └── NeuralNetworkPerturbation (parameter-vector based, no activation storage)
|
|
44
|
-
└── Optimizer (gradient computation + update rule)
|
|
45
|
-
├── OptimizerBackprop (analytical gradients via chain rule)
|
|
46
|
-
└── OptimizerPerturbation (estimated gradients via function evaluations)
|
|
47
|
-
```
|
|
48
|
-
|
|
49
|
-
**Key insight**: By treating gradients as an *estimated quantity* rather than an *exact derivative*, both methods become
|
|
50
|
-
instances of the same abstraction.
|
|
51
|
-
|
|
52
|
-
---
|
|
53
|
-
|
|
54
|
-
## Quick Start
|
|
55
|
-
|
|
56
|
-
1. **Clone the repository:**
|
|
57
|
-
```bash
|
|
58
|
-
git clone https://github.com/nicolasmalet/Zeroth-Learn.git
|
|
59
|
-
cd Zeroth-Learn
|
|
60
|
-
```
|
|
61
|
-
|
|
62
|
-
2. **Install dependencies:**
|
|
63
|
-
```bash
|
|
64
|
-
pip install -r requirements.txt
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
3. **Run a benchmark experiment:**
|
|
68
|
-
To train a linear MLP on MNIST using SPSA with 50 perturbations:
|
|
69
|
-
```bash
|
|
70
|
-
python -m lab.mnist
|
|
71
|
-
```
|
|
72
|
-
|
|
73
|
-
---
|
|
74
|
-
|
|
75
|
-
### SPSA Implementation Details
|
|
76
|
-
|
|
77
|
-
The core challenge: **evaluate multiple perturbed models in parallel without Python loops**.
|
|
78
|
-
|
|
79
|
-
#### Naive Approach (slow):
|
|
80
|
-
|
|
81
|
-
```python
|
|
82
|
-
for perturbation in perturbations:
|
|
83
|
-
theta_perturbed = theta + perturbation
|
|
84
|
-
loss_perturbed[i] = evaluate_model(theta_perturbed)
|
|
85
|
-
```
|
|
86
|
-
|
|
87
|
-
**Cost**: O(T) sequential forward passes for T perturbations.
|
|
88
|
-
|
|
89
|
-
#### Vectorized Approach (implemented):
|
|
90
|
-
|
|
91
|
-
```python
|
|
92
|
-
# Shape: (T, n_params)
|
|
93
|
-
pThetas = theta[None, :] + perturbations
|
|
94
|
-
|
|
95
|
-
# Reshape to (T, n_layers) weight matrices
|
|
96
|
-
Ws, Bs = params.from_pThetas(pThetas)
|
|
97
|
-
|
|
98
|
-
# Broadcast input across all T models simultaneously
|
|
99
|
-
# X: (batch, input_dim) -> (T, batch, input_dim)
|
|
100
|
-
for W, B, f in zip(Ws, Bs, fs):
|
|
101
|
-
X = f(X @ W + B) # Matrix multiplication broadcasts automatically
|
|
102
|
-
```
|
|
103
|
-
|
|
104
|
-
**Result**: All T forward passes execute in a single vectorized NumPy operation.
|
|
105
|
-
|
|
106
|
-
---
|
|
107
|
-
|
|
108
|
-
### Mathematical Rigor
|
|
109
|
-
|
|
110
|
-
#### Gradient Estimation
|
|
111
|
-
|
|
112
|
-
The SPSA gradient estimator:
|
|
113
|
-
|
|
114
|
-
$$\nabla L(\theta) \approx \frac{1}{T \cdot \delta} \sum_{i=1}^{T} \left( L(\theta + \delta \Delta_i) - L(\theta) \right) \Delta_i$$
|
|
115
|
-
|
|
116
|
-
where $\Delta_i \sim \text{Rademacher}(\pm 1)$ are random perturbation directions.
|
|
117
|
-
|
|
118
|
-
**Implementation** (using Einstein summation for efficiency):
|
|
119
|
-
|
|
120
|
-
```python
|
|
121
|
-
# L_diff: (T, batch_size), Ps: (T, n_params)
|
|
122
|
-
grad = np.einsum('ij,ik->k', L_diff, self.Ps) / (batch_size * T * delta)
|
|
123
|
-
```
|
|
124
|
-
|
|
125
|
-
#### Numerical Stability Considerations
|
|
126
|
-
|
|
127
|
-
- **Softmax**: Shifted by max to prevent overflow: `exp(x - max(x))`
|
|
128
|
-
- **CrossEntropy**: Added epsilon (1e-8) to prevent log(0)
|
|
129
|
-
- **Xavier initialization**: Weights sampled from $U(-\sqrt{6/(n_{in}+n_{out})}, +\sqrt{6/(n_{in}+n_{out})})$
|
|
130
|
-
|
|
131
|
-
---
|
|
132
|
-
|
|
133
|
-
## Experimental Validation
|
|
134
|
-
|
|
135
|
-
### Research Question
|
|
136
|
-
|
|
137
|
-
*What are the optimal conditions (architecture depth, learning rate, perturbation count) for SPSA to compete with
|
|
138
|
-
backpropagation?*
|
|
139
|
-
|
|
140
|
-
### Methodology
|
|
141
|
-
|
|
142
|
-
**Phase 1: Hyperparameter Sensitivity Analysis**
|
|
143
|
-
|
|
144
|
-
- Grid search over learning rates × architectures
|
|
145
|
-
- Identified stability thresholds (divergence boundaries)
|
|
146
|
-
|
|
147
|
-
<img alt="Learning Rate Analysis" src="assets/plots/lr_adam.png" height="300">
|
|
148
|
-
|
|
149
|
-
**Finding**: Adam requires lr ~ 0.001 for networks with 10K to 100K parameters to avoid gradient explosion in SPSA.
|
|
150
|
-
|
|
151
|
-
---
|
|
152
|
-
|
|
153
|
-
**Phase 2: Scalability Limits**
|
|
154
|
-
|
|
155
|
-
- Trained 6 models from 7K to 1.3M parameters (here are the first three)
|
|
156
|
-
- Measured convergence speed vs parameter count
|
|
157
|
-
|
|
158
|
-
<img alt="Architecture Scaling" src="assets/plots/small_sizes.png" height="300"/>
|
|
159
|
-
|
|
160
|
-
**Finding**: Models with 100K parameters are sufficient to get 97% accuracy
|
|
161
|
-
|
|
162
|
-
---
|
|
163
|
-
|
|
164
|
-
**Phase 3: Sample Efficiency**
|
|
165
|
-
|
|
166
|
-
- Varied perturbation count T ∈ {10, 30, 100}
|
|
167
|
-
- Measured gradient variance vs. computational cost
|
|
168
|
-
|
|
169
|
-
<img alt="Perturbation Analysis" src="assets/plots/nb_perturbations.png" height="300"/>
|
|
170
|
-
|
|
171
|
-
**Finding**: As gradient approximation variance reduction follows $\sigma \propto 1/\sqrt{T}$, we get marginal returns
|
|
172
|
-
beyond T=30.
|
|
173
|
-
|
|
174
|
-
**Practical implication**: For quantum circuits, 30 evaluations/step is feasible on current hardware.
|
|
175
|
-
|
|
176
|
-
---
|
|
177
|
-
|
|
178
|
-
## Software Engineering Practices
|
|
179
|
-
|
|
180
|
-
### Type Safety & Configuration Management
|
|
181
|
-
|
|
182
|
-
- **Frozen dataclasses** for all configs → immutable
|
|
183
|
-
- **Config serialization** → full experiment reproducibility (saved as JSON)
|
|
184
|
-
|
|
185
|
-
### Modular Design
|
|
186
|
-
|
|
187
|
-
- **Catalog pattern** for hyperparameters (see `config.py`):
|
|
188
|
-
|
|
189
|
-
```python
|
|
190
|
-
@dataclass(frozen=True)
|
|
191
|
-
class OptimizerCatalog:
|
|
192
|
-
FirstOrderAdam = FirstOrderAdamConfig(lr=0.001, ...)
|
|
193
|
-
ZerothOrderAdam = ZerothOrderAdamConfig(lr=0.001, ...)
|
|
194
|
-
```
|
|
195
|
-
|
|
196
|
-
Enables experiment generation via `itertools.product`.
|
|
197
|
-
|
|
198
|
-
### Experiment Reproducibility
|
|
199
|
-
|
|
200
|
-
- Automatic result saving (loss curves, results dataframe, hyperparameter logs)
|
|
201
|
-
- Plot styling configured globally (publication-ready figures)
|
|
202
|
-
|
|
203
|
-
---
|
|
204
|
-
|
|
205
|
-
## Software Design Principles
|
|
206
|
-
|
|
207
|
-
- **Separation of Concerns**: Gradient computation (Optimizer) is decoupled from forward pass (NeuralNetwork)
|
|
208
|
-
- **Config-Driven**: All hyperparameters defined as immutable dataclasses → reproducibility
|
|
209
|
-
- **Polymorphism**: Models can swap between backprop and SPSA without code changes
|
|
210
|
-
|
|
211
|
-
---
|
|
212
|
-
|
|
213
|
-
## Skills Demonstrated
|
|
214
|
-
|
|
215
|
-
**Deep Learning Fundamentals**: Implemented backprop from scratch (no PyTorch/TensorFlow)
|
|
216
|
-
**Numerical Optimization**: SPSA, Adam, gradient estimation theory
|
|
217
|
-
**Scientific Computing**: Vectorized NumPy, broadcasting, numerical stability
|
|
218
|
-
**Software Architecture**: Abstract base classes, config management
|
|
219
|
-
**Research Methodology**: Systematic experimentation, reproducible results
|
|
220
|
-
**Mathematical Rigor**: Gradient derivations, loss functions
|
|
221
|
-
|
|
222
|
-
---
|
|
223
|
-
|
|
224
|
-
## Next Steps
|
|
225
|
-
|
|
226
|
-
### Quantum Simulation (In Progress)
|
|
227
|
-
|
|
228
|
-
- Implement `QuantumCircuitSimulator` class using Dynamics
|
|
229
|
-
- Test SPSA on parameterized quantum circuits
|
|
230
|
-
- Validate that convergence behavior matches classical benchmarks
|
|
231
|
-
|
|
232
|
-
---
|
|
233
|
-
|
|
234
|
-
## Technical Stack
|
|
235
|
-
|
|
236
|
-
**Language**: Python
|
|
237
|
-
**Core Libraries**: NumPy (vectorization), Pandas (results), Matplotlib (visualization)
|
|
238
|
-
**Design Patterns**: Strategy (Optimizer), Abstract Factory (Config instantiation), Template Method (Model training
|
|
239
|
-
loop)
|
|
240
|
-
|
|
241
|
-
---
|
|
242
|
-
|
|
243
|
-
## Project Structure
|
|
244
|
-
|
|
245
|
-
```
|
|
246
|
-
zeroth/
|
|
247
|
-
│
|
|
248
|
-
│── first-order/ # Analytical gradient methods
|
|
249
|
-
│ ├── layer.py # Forward/backward pass logic
|
|
250
|
-
│ └── optimizers.py # SGD, Adam implementations
|
|
251
|
-
│── zeroth-order/ # Zeroth-order methods
|
|
252
|
-
│ ├── gradient_estimator.py # Gradient estimation strategies
|
|
253
|
-
│ ├── parameter_manager.py # Parameter vector management
|
|
254
|
-
│ └── optimizers.py # SPSA + Adam/SGD variants
|
|
255
|
-
│── abstract/ # Shared abstractions
|
|
256
|
-
│ ├── neural_network.py # Abstract base class
|
|
257
|
-
│ └── optimizer.py # Optimizer interface
|
|
258
|
-
└── experiment.py # Experiment orchestration
|
|
259
|
-
|
|
260
|
-
lab/
|
|
261
|
-
│
|
|
262
|
-
├── experiments.py # Pre-configured experiments
|
|
263
|
-
├── models.py # Model definitions
|
|
264
|
-
└── config.py # Hyperparameter catalogs
|
|
265
|
-
```
|
|
266
|
-
|
|
267
|
-
---
|
|
268
|
-
|
|
269
|
-
## Contact
|
|
270
|
-
|
|
271
|
-
**Nicolas Malet**
|
|
272
|
-
X24 — École Polytechnique
|
|
273
|
-
nicolas.malet@polytechnique.edu
|
|
274
|
-
[GitHub](https://github.com/nicolasmalet) | [LinkedIn](https://www.linkedin.com/in/nicolas-malet-pro)
|
|
1
|
+
# Zeroth-Learn
|
|
2
|
+
|
|
3
|
+
**A research library for zeroth-order optimization (gradient-free) in machine learning, with applications to quantum
|
|
4
|
+
computing.**
|
|
5
|
+
|
|
6
|
+
*Research project — Nicolas, X24*
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
## Context & Motivation
|
|
11
|
+
|
|
12
|
+
This project originated from a fundamental question in quantum machine learning:
|
|
13
|
+
|
|
14
|
+
> **How do you train parameterized quantum circuits when backpropagation is impossible?**
|
|
15
|
+
|
|
16
|
+
Quantum circuits are too complex to differentiate we treat it as a black box, thus we need to find alternatives to
|
|
17
|
+
backpropagation, SPSA is an excellent candidate.
|
|
18
|
+
|
|
19
|
+
**SPSA (Simultaneous Perturbation Stochastic Approximation)** solves this by estimating gradients from only O(1)
|
|
20
|
+
evaluations per iteration.
|
|
21
|
+
|
|
22
|
+
Before deploying on quantum simulators, I built this library to:
|
|
23
|
+
|
|
24
|
+
1. Understand the theoretical foundations of zeroth-order optimization
|
|
25
|
+
2. Validate SPSA stability on classical benchmarks (MNIST)
|
|
26
|
+
3. Validate my results with backpropagation models.
|
|
27
|
+
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
## Technical Implementation
|
|
31
|
+
|
|
32
|
+
### Architecture Decisions
|
|
33
|
+
|
|
34
|
+
**Problem**: Standard deep learning frameworks (PyTorch, JAX) are tightly coupled to automatic differentiation. I needed
|
|
35
|
+
an architecture where gradient computation is a **swappable abstraction**.
|
|
36
|
+
|
|
37
|
+
**Solution**: Clean separation of concerns using abstract base classes:
|
|
38
|
+
|
|
39
|
+
```
|
|
40
|
+
Model (training loop orchestration)
|
|
41
|
+
├── NeuralNetwork (forward pass interface)
|
|
42
|
+
│ ├── NeuralNetworkBackpropagation (layer-based, stores activations)
|
|
43
|
+
│ └── NeuralNetworkPerturbation (parameter-vector based, no activation storage)
|
|
44
|
+
└── Optimizer (gradient computation + update rule)
|
|
45
|
+
├── OptimizerBackprop (analytical gradients via chain rule)
|
|
46
|
+
└── OptimizerPerturbation (estimated gradients via function evaluations)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
**Key insight**: By treating gradients as an *estimated quantity* rather than an *exact derivative*, both methods become
|
|
50
|
+
instances of the same abstraction.
|
|
51
|
+
|
|
52
|
+
---
|
|
53
|
+
|
|
54
|
+
## Quick Start
|
|
55
|
+
|
|
56
|
+
1. **Clone the repository:**
|
|
57
|
+
```bash
|
|
58
|
+
git clone https://github.com/nicolasmalet/Zeroth-Learn.git
|
|
59
|
+
cd Zeroth-Learn
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
2. **Install dependencies:**
|
|
63
|
+
```bash
|
|
64
|
+
pip install -r requirements.txt
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
3. **Run a benchmark experiment:**
|
|
68
|
+
To train a linear MLP on MNIST using SPSA with 50 perturbations:
|
|
69
|
+
```bash
|
|
70
|
+
python -m lab.mnist
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
75
|
+
### SPSA Implementation Details
|
|
76
|
+
|
|
77
|
+
The core challenge: **evaluate multiple perturbed models in parallel without Python loops**.
|
|
78
|
+
|
|
79
|
+
#### Naive Approach (slow):
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
for perturbation in perturbations:
|
|
83
|
+
theta_perturbed = theta + perturbation
|
|
84
|
+
loss_perturbed[i] = evaluate_model(theta_perturbed)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
**Cost**: O(T) sequential forward passes for T perturbations.
|
|
88
|
+
|
|
89
|
+
#### Vectorized Approach (implemented):
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
# Shape: (T, n_params)
|
|
93
|
+
pThetas = theta[None, :] + perturbations
|
|
94
|
+
|
|
95
|
+
# Reshape to (T, n_layers) weight matrices
|
|
96
|
+
Ws, Bs = params.from_pThetas(pThetas)
|
|
97
|
+
|
|
98
|
+
# Broadcast input across all T models simultaneously
|
|
99
|
+
# X: (batch, input_dim) -> (T, batch, input_dim)
|
|
100
|
+
for W, B, f in zip(Ws, Bs, fs):
|
|
101
|
+
X = f(X @ W + B) # Matrix multiplication broadcasts automatically
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
**Result**: All T forward passes execute in a single vectorized NumPy operation.
|
|
105
|
+
|
|
106
|
+
---
|
|
107
|
+
|
|
108
|
+
### Mathematical Rigor
|
|
109
|
+
|
|
110
|
+
#### Gradient Estimation
|
|
111
|
+
|
|
112
|
+
The SPSA gradient estimator:
|
|
113
|
+
|
|
114
|
+
$$\nabla L(\theta) \approx \frac{1}{T \cdot \delta} \sum_{i=1}^{T} \left( L(\theta + \delta \Delta_i) - L(\theta) \right) \Delta_i$$
|
|
115
|
+
|
|
116
|
+
where $\Delta_i \sim \text{Rademacher}(\pm 1)$ are random perturbation directions.
|
|
117
|
+
|
|
118
|
+
**Implementation** (using Einstein summation for efficiency):
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
# L_diff: (T, batch_size), Ps: (T, n_params)
|
|
122
|
+
grad = np.einsum('ij,ik->k', L_diff, self.Ps) / (batch_size * T * delta)
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
#### Numerical Stability Considerations
|
|
126
|
+
|
|
127
|
+
- **Softmax**: Shifted by max to prevent overflow: `exp(x - max(x))`
|
|
128
|
+
- **CrossEntropy**: Added epsilon (1e-8) to prevent log(0)
|
|
129
|
+
- **Xavier initialization**: Weights sampled from $U(-\sqrt{6/(n_{in}+n_{out})}, +\sqrt{6/(n_{in}+n_{out})})$
|
|
130
|
+
|
|
131
|
+
---
|
|
132
|
+
|
|
133
|
+
## Experimental Validation
|
|
134
|
+
|
|
135
|
+
### Research Question
|
|
136
|
+
|
|
137
|
+
*What are the optimal conditions (architecture depth, learning rate, perturbation count) for SPSA to compete with
|
|
138
|
+
backpropagation?*
|
|
139
|
+
|
|
140
|
+
### Methodology
|
|
141
|
+
|
|
142
|
+
**Phase 1: Hyperparameter Sensitivity Analysis**
|
|
143
|
+
|
|
144
|
+
- Grid search over learning rates × architectures
|
|
145
|
+
- Identified stability thresholds (divergence boundaries)
|
|
146
|
+
|
|
147
|
+
<img alt="Learning Rate Analysis" src="assets/plots/lr_adam.png" height="300">
|
|
148
|
+
|
|
149
|
+
**Finding**: Adam requires lr ~ 0.001 for networks with 10K to 100K parameters to avoid gradient explosion in SPSA.
|
|
150
|
+
|
|
151
|
+
---
|
|
152
|
+
|
|
153
|
+
**Phase 2: Scalability Limits**
|
|
154
|
+
|
|
155
|
+
- Trained 6 models from 7K to 1.3M parameters (here are the first three)
|
|
156
|
+
- Measured convergence speed vs parameter count
|
|
157
|
+
|
|
158
|
+
<img alt="Architecture Scaling" src="assets/plots/small_sizes.png" height="300"/>
|
|
159
|
+
|
|
160
|
+
**Finding**: Models with 100K parameters are sufficient to get 97% accuracy
|
|
161
|
+
|
|
162
|
+
---
|
|
163
|
+
|
|
164
|
+
**Phase 3: Sample Efficiency**
|
|
165
|
+
|
|
166
|
+
- Varied perturbation count T ∈ {10, 30, 100}
|
|
167
|
+
- Measured gradient variance vs. computational cost
|
|
168
|
+
|
|
169
|
+
<img alt="Perturbation Analysis" src="assets/plots/nb_perturbations.png" height="300"/>
|
|
170
|
+
|
|
171
|
+
**Finding**: As gradient approximation variance reduction follows $\sigma \propto 1/\sqrt{T}$, we get marginal returns
|
|
172
|
+
beyond T=30.
|
|
173
|
+
|
|
174
|
+
**Practical implication**: For quantum circuits, 30 evaluations/step is feasible on current hardware.
|
|
175
|
+
|
|
176
|
+
---
|
|
177
|
+
|
|
178
|
+
## Software Engineering Practices
|
|
179
|
+
|
|
180
|
+
### Type Safety & Configuration Management
|
|
181
|
+
|
|
182
|
+
- **Frozen dataclasses** for all configs → immutable
|
|
183
|
+
- **Config serialization** → full experiment reproducibility (saved as JSON)
|
|
184
|
+
|
|
185
|
+
### Modular Design
|
|
186
|
+
|
|
187
|
+
- **Catalog pattern** for hyperparameters (see `config.py`):
|
|
188
|
+
|
|
189
|
+
```python
|
|
190
|
+
@dataclass(frozen=True)
|
|
191
|
+
class OptimizerCatalog:
|
|
192
|
+
FirstOrderAdam = FirstOrderAdamConfig(lr=0.001, ...)
|
|
193
|
+
ZerothOrderAdam = ZerothOrderAdamConfig(lr=0.001, ...)
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
Enables experiment generation via `itertools.product`.
|
|
197
|
+
|
|
198
|
+
### Experiment Reproducibility
|
|
199
|
+
|
|
200
|
+
- Automatic result saving (loss curves, results dataframe, hyperparameter logs)
|
|
201
|
+
- Plot styling configured globally (publication-ready figures)
|
|
202
|
+
|
|
203
|
+
---
|
|
204
|
+
|
|
205
|
+
## Software Design Principles
|
|
206
|
+
|
|
207
|
+
- **Separation of Concerns**: Gradient computation (Optimizer) is decoupled from forward pass (NeuralNetwork)
|
|
208
|
+
- **Config-Driven**: All hyperparameters defined as immutable dataclasses → reproducibility
|
|
209
|
+
- **Polymorphism**: Models can swap between backprop and SPSA without code changes
|
|
210
|
+
|
|
211
|
+
---
|
|
212
|
+
|
|
213
|
+
## Skills Demonstrated
|
|
214
|
+
|
|
215
|
+
**Deep Learning Fundamentals**: Implemented backprop from scratch (no PyTorch/TensorFlow)
|
|
216
|
+
**Numerical Optimization**: SPSA, Adam, gradient estimation theory
|
|
217
|
+
**Scientific Computing**: Vectorized NumPy, broadcasting, numerical stability
|
|
218
|
+
**Software Architecture**: Abstract base classes, config management
|
|
219
|
+
**Research Methodology**: Systematic experimentation, reproducible results
|
|
220
|
+
**Mathematical Rigor**: Gradient derivations, loss functions
|
|
221
|
+
|
|
222
|
+
---
|
|
223
|
+
|
|
224
|
+
## Next Steps
|
|
225
|
+
|
|
226
|
+
### Quantum Simulation (In Progress)
|
|
227
|
+
|
|
228
|
+
- Implement `QuantumCircuitSimulator` class using Dynamics
|
|
229
|
+
- Test SPSA on parameterized quantum circuits
|
|
230
|
+
- Validate that convergence behavior matches classical benchmarks
|
|
231
|
+
|
|
232
|
+
---
|
|
233
|
+
|
|
234
|
+
## Technical Stack
|
|
235
|
+
|
|
236
|
+
**Language**: Python
|
|
237
|
+
**Core Libraries**: NumPy (vectorization), Pandas (results), Matplotlib (visualization)
|
|
238
|
+
**Design Patterns**: Strategy (Optimizer), Abstract Factory (Config instantiation), Template Method (Model training
|
|
239
|
+
loop)
|
|
240
|
+
|
|
241
|
+
---
|
|
242
|
+
|
|
243
|
+
## Project Structure
|
|
244
|
+
|
|
245
|
+
```
|
|
246
|
+
zeroth/
|
|
247
|
+
│
|
|
248
|
+
│── first-order/ # Analytical gradient methods
|
|
249
|
+
│ ├── layer.py # Forward/backward pass logic
|
|
250
|
+
│ └── optimizers.py # SGD, Adam implementations
|
|
251
|
+
│── zeroth-order/ # Zeroth-order methods
|
|
252
|
+
│ ├── gradient_estimator.py # Gradient estimation strategies
|
|
253
|
+
│ ├── parameter_manager.py # Parameter vector management
|
|
254
|
+
│ └── optimizers.py # SPSA + Adam/SGD variants
|
|
255
|
+
│── abstract/ # Shared abstractions
|
|
256
|
+
│ ├── neural_network.py # Abstract base class
|
|
257
|
+
│ └── optimizer.py # Optimizer interface
|
|
258
|
+
└── experiment.py # Experiment orchestration
|
|
259
|
+
|
|
260
|
+
lab/
|
|
261
|
+
│
|
|
262
|
+
├── experiments.py # Pre-configured experiments
|
|
263
|
+
├── models.py # Model definitions
|
|
264
|
+
└── config.py # Hyperparameter catalogs
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
---
|
|
268
|
+
|
|
269
|
+
## Contact
|
|
270
|
+
|
|
271
|
+
**Nicolas Malet**
|
|
272
|
+
X24 — École Polytechnique
|
|
273
|
+
nicolas.malet@polytechnique.edu
|
|
274
|
+
[GitHub](https://github.com/nicolasmalet) | [LinkedIn](https://www.linkedin.com/in/nicolas-malet-pro)
|
|
@@ -1,18 +1,18 @@
|
|
|
1
|
-
[build-system]
|
|
2
|
-
requires = ["setuptools>=61.0"]
|
|
3
|
-
build-backend = "setuptools.build_meta"
|
|
4
|
-
|
|
5
|
-
[project]
|
|
6
|
-
name = "zeroth-learn"
|
|
7
|
-
version = "0.2.
|
|
8
|
-
dependencies = [
|
|
9
|
-
"numpy",
|
|
10
|
-
"pandas",
|
|
11
|
-
"matplotlib",
|
|
12
|
-
"cycler"
|
|
13
|
-
]
|
|
14
|
-
|
|
15
|
-
[tool.setuptools.packages.find]
|
|
16
|
-
where = ["."]
|
|
17
|
-
include = ["zeroth*"]
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "zeroth-learn"
|
|
7
|
+
version = "0.2.2"
|
|
8
|
+
dependencies = [
|
|
9
|
+
"numpy",
|
|
10
|
+
"pandas",
|
|
11
|
+
"matplotlib",
|
|
12
|
+
"cycler"
|
|
13
|
+
]
|
|
14
|
+
|
|
15
|
+
[tool.setuptools.packages.find]
|
|
16
|
+
where = ["."]
|
|
17
|
+
include = ["zeroth*"]
|
|
18
18
|
exclude = ["lab*"]
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
[egg_info]
|
|
2
|
-
tag_build =
|
|
3
|
-
tag_date = 0
|
|
4
|
-
|
|
1
|
+
[egg_info]
|
|
2
|
+
tag_build =
|
|
3
|
+
tag_date = 0
|
|
4
|
+
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
from .activation import Activation
|
|
2
|
-
from .blackbox import BlackBox
|
|
3
|
-
from .data_creator import DataCreator
|
|
4
|
-
from .loss import Loss
|
|
5
|
-
from .metric import Metric
|
|
6
|
-
from .model import Model, ModelConfig
|
|
7
|
-
from .neural_network import NeuralNetworkConfig,
|
|
8
|
-
from .optimizer import Optimizer
|
|
9
|
-
from .perturbation_matrix import PerturbationMatrix
|
|
10
|
-
from .summary import Summary
|
|
1
|
+
from .activation import Activation
|
|
2
|
+
from .blackbox import BlackBox
|
|
3
|
+
from .data_creator import DataCreator
|
|
4
|
+
from .loss import Loss
|
|
5
|
+
from .metric import Metric
|
|
6
|
+
from .model import Model, ModelConfig
|
|
7
|
+
from .neural_network import NeuralNetworkConfig, NetworkArchitecture, NeuralNetwork
|
|
8
|
+
from .optimizer import Optimizer
|
|
9
|
+
from .perturbation_matrix import PerturbationMatrix
|
|
10
|
+
from .summary import Summary
|
|
@@ -1,20 +1,20 @@
|
|
|
1
|
-
from abc import ABC, abstractmethod
|
|
2
|
-
|
|
3
|
-
from ..types import Array
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
class Activation(ABC):
|
|
7
|
-
"""Classe de base pour toutes les fonctions d'activation."""
|
|
8
|
-
|
|
9
|
-
def __repr__(self) -> str:
|
|
10
|
-
return f"{self.__class__.__name__}()"
|
|
11
|
-
|
|
12
|
-
@abstractmethod
|
|
13
|
-
def __call__(self, x: float | Array) -> float | Array:
|
|
14
|
-
"""Applique la fonction d'activation (forward)."""
|
|
15
|
-
...
|
|
16
|
-
|
|
17
|
-
@abstractmethod
|
|
18
|
-
def derivative(self, x: float | Array) -> float | Array:
|
|
19
|
-
"""Calcule la dérivée de la fonction d'activation."""
|
|
20
|
-
...
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
|
|
3
|
+
from ..types import Array
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Activation(ABC):
|
|
7
|
+
"""Classe de base pour toutes les fonctions d'activation."""
|
|
8
|
+
|
|
9
|
+
def __repr__(self) -> str:
|
|
10
|
+
return f"{self.__class__.__name__}()"
|
|
11
|
+
|
|
12
|
+
@abstractmethod
|
|
13
|
+
def __call__(self, x: float | Array) -> float | Array:
|
|
14
|
+
"""Applique la fonction d'activation (forward)."""
|
|
15
|
+
...
|
|
16
|
+
|
|
17
|
+
@abstractmethod
|
|
18
|
+
def derivative(self, x: float | Array) -> float | Array:
|
|
19
|
+
"""Calcule la dérivée de la fonction d'activation."""
|
|
20
|
+
...
|