zeroth-learn 0.2.4__tar.gz → 0.2.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. zeroth_learn-0.2.6/PKG-INFO +92 -0
  2. zeroth_learn-0.2.6/README.md +78 -0
  3. zeroth_learn-0.2.6/pyproject.toml +38 -0
  4. zeroth_learn-0.2.6/tests/test_gradient_estimator.py +23 -0
  5. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/__init__.py +1 -1
  6. zeroth_learn-0.2.6/zeroth/abstract/activation.py +20 -0
  7. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/blackbox.py +3 -2
  8. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/metric.py +1 -1
  9. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/model.py +16 -13
  10. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/neural_network.py +2 -2
  11. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/perturbation_matrix.py +2 -2
  12. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/summary.py +12 -7
  13. zeroth_learn-0.2.6/zeroth/all.py +5 -0
  14. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/data.py +2 -2
  15. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/experiment.py +14 -13
  16. zeroth_learn-0.2.6/zeroth/first_order/__init__.py +4 -0
  17. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/model.py +5 -6
  18. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/neural_network.py +6 -4
  19. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/optimizers.py +3 -3
  20. zeroth_learn-0.2.6/zeroth/losses/binary_cross_entropy.py +38 -0
  21. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/plot_losses.py +9 -23
  22. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/activation_functions.py +10 -11
  23. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/dataclasses_utils.py +5 -6
  24. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/metrics.py +1 -1
  25. zeroth_learn-0.2.6/zeroth/zeroth_order/__init__.py +12 -0
  26. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/gradient_estimators.py +2 -2
  27. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/model.py +5 -6
  28. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/neural_network/neural_network.py +7 -5
  29. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/neural_network/parameter_manager.py +3 -5
  30. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/optimizers.py +7 -7
  31. zeroth_learn-0.2.6/zeroth_learn.egg-info/PKG-INFO +92 -0
  32. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/SOURCES.txt +2 -0
  33. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/requires.txt +3 -0
  34. zeroth_learn-0.2.4/PKG-INFO +0 -7
  35. zeroth_learn-0.2.4/README.md +0 -274
  36. zeroth_learn-0.2.4/pyproject.toml +0 -18
  37. zeroth_learn-0.2.4/zeroth/abstract/activation.py +0 -20
  38. zeroth_learn-0.2.4/zeroth/all.py +0 -12
  39. zeroth_learn-0.2.4/zeroth/first_order/__init__.py +0 -5
  40. zeroth_learn-0.2.4/zeroth/zeroth_order/__init__.py +0 -6
  41. zeroth_learn-0.2.4/zeroth_learn.egg-info/PKG-INFO +0 -7
  42. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/MANIFEST.in +0 -0
  43. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/setup.cfg +0 -0
  44. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/__init__.py +0 -0
  45. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/data_creator.py +0 -0
  46. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/loss.py +0 -0
  47. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/abstract/optimizer.py +0 -0
  48. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/first_order/layer.py +0 -0
  49. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/losses/__init__.py +0 -0
  50. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/losses/cross_entropy.py +0 -0
  51. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/losses/mse.py +0 -0
  52. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/paths.py +0 -0
  53. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/types.py +0 -0
  54. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/__init__.py +0 -0
  55. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/utils/perturbation_matrices.py +0 -0
  56. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/neural_network/__init__.py +0 -0
  57. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth/zeroth_order/zeroth_order_blackbox.py +1 -1
  58. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/dependency_links.txt +0 -0
  59. {zeroth_learn-0.2.4 → zeroth_learn-0.2.6}/zeroth_learn.egg-info/top_level.txt +0 -0
@@ -0,0 +1,92 @@
1
+ Metadata-Version: 2.4
2
+ Name: zeroth-learn
3
+ Version: 0.2.6
4
+ Summary: Zeroth-order optimization for black-box models
5
+ Project-URL: Repository, https://github.com/nicolasmalet/Zeroth-Learn
6
+ Requires-Python: >=3.11
7
+ Description-Content-Type: text/markdown
8
+ Requires-Dist: numpy
9
+ Requires-Dist: pandas
10
+ Requires-Dist: matplotlib
11
+ Requires-Dist: cycler
12
+ Provides-Extra: mnist
13
+ Requires-Dist: scikit-learn; extra == "mnist"
14
+
15
+ # Zeroth-Learn
16
+
17
+ **A NumPy library for estimating gradients from loss evaluations and training black-box models.**
18
+
19
+ The library implements coordinate finite differences, random-direction estimation, and SGD/Adam updates. Its MNIST experiments check that these components can train models when backpropagation is unavailable. The estimator and optimizer interfaces can also be used with other black boxes, including the simulator in [Quantum-Learn](https://github.com/nicolasmalet/Quantum-Learn).
20
+
21
+ ## 1. Optimisation without derivatives
22
+
23
+ Let $\theta\in\mathbb R^d$ be model parameters and $f(\theta)$ a mini-batch loss. The model exposes values of $f$, but not $\nabla f$. Zeroth-order optimisation estimates a descent direction by evaluating nearby parameters:
24
+
25
+ ```math
26
+ \theta_{k+1}=\theta_k-\eta_k\widehat{\nabla f}(\theta_k).
27
+ ```
28
+
29
+ For Adam, the estimated gradient enters the usual first- and second-moment update in place of an exact gradient. No derivative of the black-box model is required.
30
+
31
+ ## 2. Finite-difference estimators
32
+
33
+ A one-sided coordinate difference estimates component $i$ as
34
+
35
+ ```math
36
+ \widehat{\partial_i f}(\theta)
37
+ =\frac{f(\theta+\delta e_i)-f(\theta)}{\delta}.
38
+ ```
39
+
40
+ Estimating all $d$ components needs $d+1$ loss evaluations. [`GlobalFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) does this for every coordinate; [`PartialFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) evaluates only selected coordinates and returns zero elsewhere.
41
+
42
+ For a cheaper estimate in high dimension, [`SimultaneousPerturbation`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) draws $T$ Rademacher directions. At construction, the matrix $P\in\mathbb R^{T\times d}$ has entries $P_{ij}=s_{ij}/\sqrt T$, where each $s_{ij}$ is $+1$ or $-1$ with equal probability. The implementation uses
43
+
44
+ ```math
45
+ \widehat{\nabla f}(\theta)
46
+ =\frac{1}{\delta}P^\top
47
+ \begin{pmatrix}
48
+ f(\theta+\delta P_1)-f(\theta)\\
49
+ \vdots\\
50
+ f(\theta+\delta P_T)-f(\theta)
51
+ \end{pmatrix}.
52
+ ```
53
+
54
+ Because $\mathbb E[P^\top P]=I$, the first-order term recovers $\nabla f$ in expectation. Each update costs $T+1$ loss evaluations, independent of $d$ for a fixed $T$. Larger $T$ averages more directions but increases evaluation cost and memory.
55
+
56
+ ## 3. From an estimator to a model update
57
+
58
+ The interfaces separate three operations:
59
+
60
+ | Operation | Implementation |
61
+ | ---------------------------------------------------------- | --------------------------------------------------------------------------- |
62
+ | Generate perturbed parameters and reconstruct the gradient | [`gradient_estimators.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) |
63
+ | Generate Rademacher directions | [`perturbation_matrices.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/utils/perturbation_matrices.py) |
64
+ | Evaluate a model at nominal and perturbed parameters | [`zeroth_order_blackbox.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/zeroth_order_blackbox.py) |
65
+ | Apply SGD or Adam to the estimated gradient | [`optimizers.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/optimizers.py) |
66
+
67
+ The included [`ZerothOrderNeuralNetwork`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/neural_network.py) is one black-box implementation. [`ParameterManager`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/parameter_manager.py) maps weights and biases to a flat $\theta$; NumPy broadcasting evaluates its $T+1$ perturbed networks in one batch. This batching reduces Python overhead, not the number of loss evaluations. Other models choose how to perform those evaluations.
68
+
69
+ ## 4. Numerical checks
70
+
71
+ A deterministic [test](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/tests/test_gradient_estimator.py) applies the random-direction estimator to a linear function, whose gradient is known. The MNIST experiments provide end-to-end checks of zeroth-order training on CPU. Their protocols, results, plots, and raw loss traces are in [Experiments](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/EXPERIMENTS.md).
72
+
73
+ ## 5. Reproduce the experiments
74
+
75
+ With [uv](https://docs.astral.sh/uv/), the lockfile fixes the dependency versions. The first experiment run downloads MNIST from OpenML.
76
+
77
+ ```bash
78
+ git clone https://github.com/nicolasmalet/Zeroth-Learn.git
79
+ cd Zeroth-Learn
80
+ uv sync --locked --extra mnist
81
+ uv run --locked --extra mnist python -m unittest discover -s tests
82
+ MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist
83
+ MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist.run_experiment nb_perturbations_vs_model_size
84
+ ```
85
+
86
+ The experiments write accuracy, loss traces, and plots under `results/`.
87
+
88
+ Without uv, install the package with its MNIST extra in a virtual environment, then run the same Python modules.
89
+
90
+ ## 6. Scope
91
+
92
+ The MNIST experiments validate classical models. They are not benchmarks of quantum hardware or of PyTorch, and their saved plots do not contain repeated trials or uncertainty estimates. Memory use grows with the number of perturbations in the vectorized neural-network implementation.
@@ -0,0 +1,78 @@
1
+ # Zeroth-Learn
2
+
3
+ **A NumPy library for estimating gradients from loss evaluations and training black-box models.**
4
+
5
+ The library implements coordinate finite differences, random-direction estimation, and SGD/Adam updates. Its MNIST experiments check that these components can train models when backpropagation is unavailable. The estimator and optimizer interfaces can also be used with other black boxes, including the simulator in [Quantum-Learn](https://github.com/nicolasmalet/Quantum-Learn).
6
+
7
+ ## 1. Optimisation without derivatives
8
+
9
+ Let $\theta\in\mathbb R^d$ be model parameters and $f(\theta)$ a mini-batch loss. The model exposes values of $f$, but not $\nabla f$. Zeroth-order optimisation estimates a descent direction by evaluating nearby parameters:
10
+
11
+ ```math
12
+ \theta_{k+1}=\theta_k-\eta_k\widehat{\nabla f}(\theta_k).
13
+ ```
14
+
15
+ For Adam, the estimated gradient enters the usual first- and second-moment update in place of an exact gradient. No derivative of the black-box model is required.
16
+
17
+ ## 2. Finite-difference estimators
18
+
19
+ A one-sided coordinate difference estimates component $i$ as
20
+
21
+ ```math
22
+ \widehat{\partial_i f}(\theta)
23
+ =\frac{f(\theta+\delta e_i)-f(\theta)}{\delta}.
24
+ ```
25
+
26
+ Estimating all $d$ components needs $d+1$ loss evaluations. [`GlobalFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) does this for every coordinate; [`PartialFiniteDifference`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) evaluates only selected coordinates and returns zero elsewhere.
27
+
28
+ For a cheaper estimate in high dimension, [`SimultaneousPerturbation`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) draws $T$ Rademacher directions. At construction, the matrix $P\in\mathbb R^{T\times d}$ has entries $P_{ij}=s_{ij}/\sqrt T$, where each $s_{ij}$ is $+1$ or $-1$ with equal probability. The implementation uses
29
+
30
+ ```math
31
+ \widehat{\nabla f}(\theta)
32
+ =\frac{1}{\delta}P^\top
33
+ \begin{pmatrix}
34
+ f(\theta+\delta P_1)-f(\theta)\\
35
+ \vdots\\
36
+ f(\theta+\delta P_T)-f(\theta)
37
+ \end{pmatrix}.
38
+ ```
39
+
40
+ Because $\mathbb E[P^\top P]=I$, the first-order term recovers $\nabla f$ in expectation. Each update costs $T+1$ loss evaluations, independent of $d$ for a fixed $T$. Larger $T$ averages more directions but increases evaluation cost and memory.
41
+
42
+ ## 3. From an estimator to a model update
43
+
44
+ The interfaces separate three operations:
45
+
46
+ | Operation | Implementation |
47
+ | ---------------------------------------------------------- | --------------------------------------------------------------------------- |
48
+ | Generate perturbed parameters and reconstruct the gradient | [`gradient_estimators.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/gradient_estimators.py) |
49
+ | Generate Rademacher directions | [`perturbation_matrices.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/utils/perturbation_matrices.py) |
50
+ | Evaluate a model at nominal and perturbed parameters | [`zeroth_order_blackbox.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/zeroth_order_blackbox.py) |
51
+ | Apply SGD or Adam to the estimated gradient | [`optimizers.py`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/optimizers.py) |
52
+
53
+ The included [`ZerothOrderNeuralNetwork`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/neural_network.py) is one black-box implementation. [`ParameterManager`](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/zeroth/zeroth_order/neural_network/parameter_manager.py) maps weights and biases to a flat $\theta$; NumPy broadcasting evaluates its $T+1$ perturbed networks in one batch. This batching reduces Python overhead, not the number of loss evaluations. Other models choose how to perform those evaluations.
54
+
55
+ ## 4. Numerical checks
56
+
57
+ A deterministic [test](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/tests/test_gradient_estimator.py) applies the random-direction estimator to a linear function, whose gradient is known. The MNIST experiments provide end-to-end checks of zeroth-order training on CPU. Their protocols, results, plots, and raw loss traces are in [Experiments](https://github.com/nicolasmalet/Zeroth-Learn/blob/main/EXPERIMENTS.md).
58
+
59
+ ## 5. Reproduce the experiments
60
+
61
+ With [uv](https://docs.astral.sh/uv/), the lockfile fixes the dependency versions. The first experiment run downloads MNIST from OpenML.
62
+
63
+ ```bash
64
+ git clone https://github.com/nicolasmalet/Zeroth-Learn.git
65
+ cd Zeroth-Learn
66
+ uv sync --locked --extra mnist
67
+ uv run --locked --extra mnist python -m unittest discover -s tests
68
+ MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist
69
+ MPLBACKEND=Agg uv run --locked --extra mnist python -m lab.mnist.run_experiment nb_perturbations_vs_model_size
70
+ ```
71
+
72
+ The experiments write accuracy, loss traces, and plots under `results/`.
73
+
74
+ Without uv, install the package with its MNIST extra in a virtual environment, then run the same Python modules.
75
+
76
+ ## 6. Scope
77
+
78
+ The MNIST experiments validate classical models. They are not benchmarks of quantum hardware or of PyTorch, and their saved plots do not contain repeated trials or uncertainty estimates. Memory use grows with the number of perturbations in the vectorized neural-network implementation.
@@ -0,0 +1,38 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "zeroth-learn"
7
+ version = "0.2.6"
8
+ description = "Zeroth-order optimization for black-box models"
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ dependencies = [
12
+ "numpy",
13
+ "pandas",
14
+ "matplotlib",
15
+ "cycler",
16
+ ]
17
+
18
+ [project.optional-dependencies]
19
+ mnist = ["scikit-learn"]
20
+
21
+ [project.urls]
22
+ Repository = "https://github.com/nicolasmalet/Zeroth-Learn"
23
+
24
+ [tool.setuptools.packages.find]
25
+ where = ["."]
26
+ include = ["zeroth*"]
27
+ exclude = ["lab*"]
28
+
29
+ [tool.ruff]
30
+ line-length = 120
31
+
32
+ [tool.ruff.lint]
33
+ select = ["E4", "E7", "E9", "F", "I", "UP", "B"]
34
+
35
+ [tool.ruff.lint.per-file-ignores]
36
+ "**/__init__.py" = ["F401", "F403"]
37
+ "zeroth/all.py" = ["F403"]
38
+ "lab/mnist/neural_networks.py" = ["E741"]
@@ -0,0 +1,23 @@
1
+ import unittest
2
+
3
+ import numpy as np
4
+
5
+ from zeroth.utils.perturbation_matrices import RademacherMatrix
6
+ from zeroth.zeroth_order.gradient_estimators import SimultaneousPerturbation
7
+
8
+
9
+ class SimultaneousPerturbationTest(unittest.TestCase):
10
+ def test_recovers_linear_gradient(self):
11
+ np.random.seed(42)
12
+ expected = np.array([1.5, -2.0, 0.5])
13
+ theta = np.array([0.2, -0.1, 0.3])
14
+ estimator = SimultaneousPerturbation(1e-6, 2_000, RademacherMatrix(), theta.size)
15
+
16
+ perturbed_theta = estimator.perturb(theta)
17
+ losses = (perturbed_theta @ expected)[:, None]
18
+
19
+ np.testing.assert_allclose(estimator.get_gradient(losses), expected, atol=0.06)
20
+
21
+
22
+ if __name__ == "__main__":
23
+ unittest.main()
@@ -4,7 +4,7 @@ from .data_creator import DataCreator
4
4
  from .loss import Loss
5
5
  from .metric import Metric
6
6
  from .model import Model, ModelConfig
7
- from .neural_network import NeuralNetworkConfig, NeuralNetwork
7
+ from .neural_network import NeuralNetwork, NeuralNetworkConfig
8
8
  from .optimizer import Optimizer
9
9
  from .perturbation_matrix import PerturbationMatrix
10
10
  from .summary import Summary
@@ -0,0 +1,20 @@
1
+ from abc import ABC, abstractmethod
2
+
3
+ from ..types import Array
4
+
5
+
6
+ class Activation(ABC):
7
+ """Base class for activation functions."""
8
+
9
+ def __repr__(self) -> str:
10
+ return f"{self.__class__.__name__}()"
11
+
12
+ @abstractmethod
13
+ def __call__(self, x: Array) -> Array:
14
+ """Apply the activation function."""
15
+ ...
16
+
17
+ @abstractmethod
18
+ def derivative(self, x: Array) -> Array:
19
+ """Compute the activation derivative."""
20
+ ...
@@ -1,4 +1,5 @@
1
1
  from abc import ABC, abstractmethod
2
+ from typing import Any
2
3
 
3
4
  from ..types import Array
4
5
 
@@ -24,7 +25,7 @@ class BlackBox(ABC):
24
25
  ...
25
26
 
26
27
  @abstractmethod
27
- def init_params(self, params: dict) -> None:
28
+ def init_params(self, params: dict[str, Any]) -> None:
28
29
  """Manually initializes the weights and biases of the network.
29
30
 
30
31
  Args:
@@ -33,7 +34,7 @@ class BlackBox(ABC):
33
34
  ...
34
35
 
35
36
  @abstractmethod
36
- def get_params(self) -> dict:
37
+ def get_params(self) -> dict[str, Any]:
37
38
  """Retrieves the current parameters of the network.
38
39
 
39
40
  Returns:
@@ -9,5 +9,5 @@ class Metric(ABC):
9
9
 
10
10
  @abstractmethod
11
11
  def __call__(self, Y_pred: Array, Y_true: Array) -> float:
12
- """Applique la fonction d'activation (forward)."""
12
+ """Compute the metric from predictions and targets."""
13
13
  ...
@@ -4,19 +4,21 @@ import pickle
4
4
  from abc import ABC, abstractmethod
5
5
  from dataclasses import dataclass
6
6
  from pathlib import Path
7
- from typing import Callable
7
+ from typing import Any
8
8
 
9
9
  import matplotlib.pyplot as plt
10
10
  import numpy as np
11
11
  import pandas as pd
12
+ from matplotlib.figure import Figure
12
13
 
14
+ from ..data import Data
15
+ from ..plot_losses import plot_losses
16
+ from ..types import Array
13
17
  from .blackbox import BlackBox
14
18
  from .loss import Loss
19
+ from .metric import Metric
15
20
  from .optimizer import Optimizer
16
21
  from .summary import Summary
17
- from ..data import Data
18
- from ..plot_losses import plot_losses
19
- from ..types import Array
20
22
 
21
23
 
22
24
  @dataclass(frozen=True, kw_only=True)
@@ -30,9 +32,9 @@ class ModelConfig(ABC, Summary):
30
32
  nb_epochs (int): Number of passes through the entire dataset.
31
33
  """
32
34
  name: str
33
- id: dict
35
+ id: dict[str, Any]
34
36
  loss: Loss
35
- metric: Callable
37
+ metric: Metric
36
38
  batch_size: int
37
39
  nb_epochs: int = 1
38
40
 
@@ -60,10 +62,10 @@ class Model(ABC):
60
62
  self.config = config
61
63
 
62
64
  self.name: str = config.name
63
- self.id: dict = config.id
65
+ self.id: dict[str, Any] = config.id
64
66
  self.data: Data = data
65
67
  self.loss: Loss = config.loss
66
- self.metric: Callable = config.metric
68
+ self.metric: Metric = config.metric
67
69
  self.batch_size: int = config.batch_size
68
70
  self.nb_epochs: int = config.nb_epochs
69
71
 
@@ -91,7 +93,7 @@ class Model(ABC):
91
93
  print_indexes = np.linspace(0, nb_batches - 1, nb_print).astype(int)
92
94
 
93
95
  for epoch_idx in range(self.nb_epochs):
94
- print(f" epoch n°{epoch_idx + 1} out of {self.nb_epochs}")
96
+ print(f" epoch {epoch_idx + 1} of {self.nb_epochs}")
95
97
  self.data.permutation()
96
98
  self.data.batch_size = self.batch_size
97
99
 
@@ -100,12 +102,12 @@ class Model(ABC):
100
102
  self.training_loss[epoch_idx * nb_batches + batch_idx] = avg_loss
101
103
 
102
104
  if batch_idx in print_indexes:
103
- print(f" batch n°{batch_idx + 1} out of {nb_batches}, "
105
+ print(f" batch {batch_idx + 1} of {nb_batches}, "
104
106
  f"loss : {np.round(self.training_loss[epoch_idx * nb_batches + batch_idx], 3)}")
105
107
 
106
108
  self.test()
107
109
 
108
- def plot_loss(self, smooth_fraction: float = 0.05) -> plt.Figure:
110
+ def plot_loss(self, smooth_fraction: float = 0.05) -> Figure:
109
111
  fig = plot_losses(dimension=0,
110
112
  models=[self],
111
113
  title=self.name,
@@ -113,7 +115,7 @@ class Model(ABC):
113
115
  plt.close(fig)
114
116
  return fig
115
117
 
116
- def test(self) -> None:
118
+ def test(self) -> Array:
117
119
  X_test, Y_true = self.data.X_test, self.data.Y_test # (in, batch), (out, batch)
118
120
  Y_pred = self.neural_network(X_test) # (out, batch)
119
121
 
@@ -121,6 +123,7 @@ class Model(ABC):
121
123
  self.test_loss = self.loss.compute_loss(Y_pred, Y_true)
122
124
 
123
125
  print(f" {self.id} accuracy : {self.test_accuracy}, loss : {self.test_loss}")
126
+ return Y_pred
124
127
 
125
128
  def save_loss(self, save_dir: Path) -> None:
126
129
  save_dir.mkdir(parents=True, exist_ok=True)
@@ -140,7 +143,7 @@ class Model(ABC):
140
143
  pickle.dump(params_dict, f)
141
144
 
142
145
  def load_weights(self, load_dir: Path) -> None:
143
- """Restaure les paramètres depuis un fichier pickle."""
146
+ """Restore model parameters from a pickle file."""
144
147
  load_path = load_dir / self.WEIGHTS_FILE
145
148
  with load_path.open('rb') as f:
146
149
  params_dict = pickle.load(f)
@@ -1,7 +1,7 @@
1
- from abc import ABC
2
1
  from dataclasses import dataclass
3
2
 
4
3
  from .activation import Activation
4
+ from .blackbox import BlackBox
5
5
  from .summary import Summary
6
6
 
7
7
 
@@ -12,7 +12,7 @@ class NeuralNetworkConfig(Summary):
12
12
  activations: list[Activation]
13
13
 
14
14
 
15
- class NeuralNetwork(ABC):
15
+ class NeuralNetwork(BlackBox):
16
16
  def __init__(self, config: NeuralNetworkConfig, input_dim: int, output_dim: int):
17
17
  self.name: str = config.name
18
18
  self.nb_layers: int = len(config.activations)
@@ -4,12 +4,12 @@ from ..types import Array
4
4
 
5
5
 
6
6
  class PerturbationMatrix(ABC):
7
- """Classe de base pour toutes les fonctions d'activation."""
7
+ """Base class for perturbation-matrix generators."""
8
8
 
9
9
  def __repr__(self) -> str:
10
10
  return f"{self.__class__.__name__}()"
11
11
 
12
12
  @abstractmethod
13
13
  def __call__(self, nb_perturbation: int, nb_parameters: int) -> Array:
14
- """Applique la fonction d'activation (forward)."""
14
+ """Generate perturbation directions."""
15
15
  ...
@@ -16,17 +16,20 @@ class Summary:
16
16
  print(self._summary(self, indent=0), file=f)
17
17
 
18
18
  @classmethod
19
- def load(cls, path: str, context: dict = None) -> Summary:
20
- with open(path, "r") as f:
21
- code = "".join(l for l in f)
22
- return eval(code.strip(), {"__builtins__": __builtins__}, context)
19
+ def load(cls, path: str, context: dict[str, Any] | None = None) -> Summary:
20
+ with open(path) as f:
21
+ code = "".join(line for line in f)
22
+ result = eval(code.strip(), {"__builtins__": __builtins__}, context)
23
+ if not isinstance(result, cls):
24
+ raise TypeError(f"Expected {cls.__name__}, got {type(result).__name__}")
25
+ return result
23
26
 
24
27
  def _summary(self, obj: Any, indent: int) -> str:
25
28
  shift = " " * indent
26
29
  next_shift = " " * (indent + 1)
27
30
 
28
31
  if dataclasses.is_dataclass(obj):
29
- cls_name = obj.__class__.__name__
32
+ cls_name = type(obj).__name__
30
33
  items = []
31
34
  for f in dataclasses.fields(obj):
32
35
  val = getattr(obj, f.name)
@@ -37,13 +40,15 @@ class Summary:
37
40
  return f"{cls_name}(\n{content}\n{shift})"
38
41
 
39
42
  elif isinstance(obj, list):
40
- if not obj: return "[]"
43
+ if not obj:
44
+ return "[]"
41
45
  items = [self._summary(item, indent + 1) for item in obj]
42
46
  content = ",\n".join([f"{next_shift}{item.lstrip()}" for item in items])
43
47
  return f"[\n{content}\n{shift}]"
44
48
 
45
49
  elif isinstance(obj, dict):
46
- if not obj: return "{}"
50
+ if not obj:
51
+ return "{}"
47
52
  items = [f"{next_shift}{repr(k)}: {self._summary(v, indent + 1).lstrip()}" for k, v in obj.items()]
48
53
  content = ",\n".join(items)
49
54
  return f"{{\n{content}\n{shift}}}"
@@ -0,0 +1,5 @@
1
+ from .experiment import *
2
+ from .first_order import *
3
+ from .losses import *
4
+ from .utils import *
5
+ from .zeroth_order import *
@@ -1,4 +1,4 @@
1
- from typing import Iterator
1
+ from collections.abc import Iterator
2
2
 
3
3
  import numpy as np
4
4
 
@@ -17,7 +17,7 @@ class Data:
17
17
  self.input_dim: int = self.raw_X_train.shape[1]
18
18
  self.output_dim: int = self.raw_Y_train.shape[1] if nb_class == 0 else nb_class
19
19
 
20
- self.batch_size: int | None = None
20
+ self.batch_size: int = 1
21
21
 
22
22
  self.indices: np.ndarray = np.arange(self.nb_data)
23
23
 
@@ -1,14 +1,15 @@
1
1
  from __future__ import annotations
2
2
 
3
3
  import itertools
4
- from pathlib import Path
5
4
  from dataclasses import dataclass, replace
6
- from typing import Union
5
+ from pathlib import Path
6
+ from typing import Any, cast
7
7
 
8
8
  import matplotlib.pyplot as plt
9
9
  import pandas as pd
10
+ from matplotlib.figure import Figure
10
11
 
11
- from .abstract import Model, ModelConfig, DataCreator, Summary
12
+ from .abstract import DataCreator, Model, ModelConfig, Summary
12
13
  from .data import Data
13
14
  from .plot_losses import plot_losses
14
15
  from .utils.dataclasses_utils import get_name, set_value_by_path
@@ -18,7 +19,7 @@ from .utils.dataclasses_utils import get_name, set_value_by_path
18
19
  class VariationConfig:
19
20
  name: str
20
21
  param: list[str]
21
- values: Union[list, list[list]]
22
+ values: list[list[Any]]
22
23
 
23
24
 
24
25
  @dataclass(frozen=True)
@@ -54,11 +55,11 @@ class Experiment:
54
55
  self.models: list[Model] = generate_models(config.base_model, config.variations, self.data)
55
56
 
56
57
  def train_models(self, nb_print: int) -> None:
57
- print(f"Training Models")
58
+ print("Training Models")
58
59
  for model in self.models:
59
60
  model.train(nb_print)
60
61
 
61
- def plot_losses(self, title: str, plot_dimension: int, smooth_fraction: float = 0) -> plt.Figure:
62
+ def plot_losses(self, title: str, plot_dimension: int, smooth_fraction: float = 0) -> Figure:
62
63
  fig = plot_losses(title=title,
63
64
  dimension=plot_dimension,
64
65
  models=self.models,
@@ -69,7 +70,7 @@ class Experiment:
69
70
  return fig
70
71
 
71
72
  def test_models(self) -> None:
72
- print(f"Testing Models")
73
+ print("Testing Models")
73
74
  for model in self.models:
74
75
  model.test()
75
76
 
@@ -88,7 +89,7 @@ class Experiment:
88
89
  df.to_csv(save_path)
89
90
 
90
91
  def save_weights(self, save_dir: Path) -> None:
91
- for i, model in enumerate(self.models):
92
+ for model in self.models:
92
93
  save_path = save_dir / model.name
93
94
  model.save_weights(save_path)
94
95
 
@@ -97,7 +98,7 @@ class Experiment:
97
98
  config_path = save_dir / self.CONFIG_FILE
98
99
  self.config.save(config_path)
99
100
 
100
- for i, model in enumerate(self.models):
101
+ for model in self.models:
101
102
  save_path = save_dir / model.name
102
103
  model.config.save(save_path)
103
104
 
@@ -109,13 +110,13 @@ def generate_models(base_model: ModelConfig, variations: list[VariationConfig],
109
110
 
110
111
  for combination in itertools.product(*values_lists):
111
112
  id_ = {}
112
- current_model = base_model
113
+ current_model: ModelConfig = base_model
113
114
 
114
- for var_config, current_vals in zip(variations, combination):
115
+ for var_config, current_vals in zip(variations, combination, strict=True):
115
116
  id_[var_config.name] = get_name(current_vals[0])
116
117
 
117
- for path, val in zip(var_config.param, current_vals):
118
- current_model = set_value_by_path(current_model, path, val)
118
+ for path, val in zip(var_config.param, current_vals, strict=True):
119
+ current_model = cast(ModelConfig, set_value_by_path(current_model, path, val))
119
120
 
120
121
  current_model = replace(current_model, id=id_)
121
122
  models.append(current_model.instantiate(data))
@@ -0,0 +1,4 @@
1
+ from .layer import Layer
2
+ from .model import FirstOrderModel, FirstOrderModelConfig
3
+ from .neural_network import FirstOrderNeuralNetwork
4
+ from .optimizers import FirstOrderAdamConfig, FirstOrderOptimizer, FirstOrderOptimizerConfig, FirstOrderSGDConfig
@@ -2,10 +2,10 @@ from __future__ import annotations
2
2
 
3
3
  from dataclasses import dataclass
4
4
 
5
- from .neural_network import FirstOrderNeuralNetwork
6
- from .optimizers import FirstOrderOptimizerConfig, FirstOrderOptimizer
7
- from ..abstract import ModelConfig, Model, NeuralNetworkConfig
5
+ from ..abstract import Model, ModelConfig, NeuralNetworkConfig
8
6
  from ..data import Data
7
+ from .neural_network import FirstOrderNeuralNetwork
8
+ from .optimizers import FirstOrderOptimizerConfig
9
9
 
10
10
 
11
11
  @dataclass(frozen=True, kw_only=True)
@@ -21,6 +21,5 @@ class FirstOrderModel(Model):
21
21
  def __init__(self, config: FirstOrderModelConfig, data: Data) -> None:
22
22
  super().__init__(config, data)
23
23
 
24
- self.neural_network: FirstOrderNeuralNetwork = FirstOrderNeuralNetwork(config.neural_network_config,
25
- data.input_dim, data.output_dim)
26
- self.optimizer: FirstOrderOptimizer = config.optimizer_config.instantiate()
24
+ self.neural_network = FirstOrderNeuralNetwork(config.neural_network_config, data.input_dim, data.output_dim)
25
+ self.optimizer = config.optimizer_config.instantiate()
@@ -1,6 +1,8 @@
1
- from .layer import Layer
1
+ from typing import Any
2
+
2
3
  from ..abstract.neural_network import NeuralNetwork, NeuralNetworkConfig
3
4
  from ..types import Array
5
+ from .layer import Layer
4
6
 
5
7
 
6
8
  class FirstOrderNeuralNetwork(NeuralNetwork):
@@ -25,14 +27,14 @@ class FirstOrderNeuralNetwork(NeuralNetwork):
25
27
  def __call__(self, X: Array) -> Array:
26
28
  return self.forward(X)
27
29
 
28
- def init_params(self, params: dict) -> None:
30
+ def init_params(self, params: dict[str, Any]) -> None:
29
31
  Ws = params["Ws"]
30
32
  Bs = params["Bs"]
31
- for layer, W, B in zip(self.layers, Ws, Bs):
33
+ for layer, W, B in zip(self.layers, Ws, Bs, strict=True):
32
34
  layer.W = W
33
35
  layer.B = B
34
36
 
35
- def get_params(self) -> dict:
37
+ def get_params(self) -> dict[str, Any]:
36
38
  Ws, Bs = [], []
37
39
  for layer in self.layers:
38
40
  Ws.append(layer.W)