spingalett 0.12.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spingalett-0.12.0/PKG-INFO +164 -0
- spingalett-0.12.0/README.md +146 -0
- spingalett-0.12.0/pyproject.toml +28 -0
- spingalett-0.12.0/setup.cfg +4 -0
- spingalett-0.12.0/setup.py +39 -0
- spingalett-0.12.0/spingalett/__init__.py +2012 -0
- spingalett-0.12.0/spingalett/py.typed +0 -0
- spingalett-0.12.0/spingalett.egg-info/PKG-INFO +164 -0
- spingalett-0.12.0/spingalett.egg-info/SOURCES.txt +10 -0
- spingalett-0.12.0/spingalett.egg-info/dependency_links.txt +1 -0
- spingalett-0.12.0/spingalett.egg-info/requires.txt +1 -0
- spingalett-0.12.0/spingalett.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: spingalett
|
|
3
|
+
Version: 0.12.0
|
|
4
|
+
Summary: Neural networks in C23 from Python: training, graphs, ONNX and PyTorch import, INT8 to INT2 deployment
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/pka-human/Spingalett
|
|
7
|
+
Keywords: neural networks,deep learning,inference,quantization,onnx,microcontrollers
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Programming Language :: C
|
|
10
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
11
|
+
Classifier: Operating System :: MacOS
|
|
12
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Classifier: Typing :: Typed
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Requires-Dist: numpy>=1.20
|
|
18
|
+
|
|
19
|
+
# Spingalett for Python
|
|
20
|
+
|
|
21
|
+
Thin `ctypes` bindings over the Spingalett shared library: no compiler is needed to install them,
|
|
22
|
+
only NumPy and `libspingalett` (`.so` / `.dylib` / `.dll`). The wheels on PyPI and on the GitHub
|
|
23
|
+
releases carry the library inside (Linux x86-64 and AArch64 with glibc 2.28 or newer, Windows
|
|
24
|
+
x86-64, macOS 11 and newer, for any Python 3), with OpenMP:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install spingalett
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
From a source checkout, build the library and point the bindings at it:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
cmake -S . -B Build -DCMAKE_BUILD_TYPE=Release -DBUILD_WITH_OPENMP=ON
|
|
34
|
+
cmake --build Build --parallel # produces Bin/libspingalett.so
|
|
35
|
+
pip install ./Bindings/Python
|
|
36
|
+
export SPINGALETT_LIBRARY=$PWD/Bin/libspingalett.so # or put Bin/ on the library path
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Without `SPINGALETT_LIBRARY` the package looks inside itself (wheels), then in the repository's
|
|
40
|
+
`Bin/` directory (when imported from a checkout), then on the system library path. The package is
|
|
41
|
+
typed (`py.typed`): editors and type checkers read its annotations.
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
import numpy as np
|
|
45
|
+
import spingalett as sg
|
|
46
|
+
|
|
47
|
+
x = np.array([[0, 0], [0, 1], [1, 0], [1, 1]], dtype=np.float32)
|
|
48
|
+
y = np.array([[0], [1], [1], [0]], dtype=np.float32)
|
|
49
|
+
|
|
50
|
+
sg.set_compute_mode(sg.ComputeMode.OPENBLAS) # falls back to single-threaded if unavailable
|
|
51
|
+
sg.seed(42)
|
|
52
|
+
|
|
53
|
+
layers = [
|
|
54
|
+
sg.Layer(2),
|
|
55
|
+
sg.Layer(16, sg.Activation.TANH, sg.Init.XAVIER, dropout=0.1),
|
|
56
|
+
sg.Layer(1, sg.Activation.SIGMOID, sg.Init.XAVIER),
|
|
57
|
+
]
|
|
58
|
+
with sg.Network(sg.Loss.MSE, layers) as net:
|
|
59
|
+
net.train(x, y, epochs=3000, optimizer=sg.Optimizer.ADAMW, learning_rate=0.02,
|
|
60
|
+
weight_decay=1e-4, lr_scheduler=sg.WarmupCosine(warmup_epochs=100),
|
|
61
|
+
callback=lambda net, progress: progress.train_loss < 1e-4, callback_interval=100)
|
|
62
|
+
print(net.forward(x)) # one row per sample
|
|
63
|
+
net.save("xor", precision=sg.Precision.FP16)
|
|
64
|
+
|
|
65
|
+
# validation, early stopping and the best epoch's weights
|
|
66
|
+
x_train, y_train = sg.load_idx("train-images-idx3-ubyte", "train-labels-idx1-ubyte", num_classes=10)
|
|
67
|
+
with sg.Network(sg.Loss.CROSS_ENTROPY, [784, sg.Layer(128, sg.Activation.RELU, sg.Init.HE),
|
|
68
|
+
sg.Layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER)]) as net:
|
|
69
|
+
result = net.train(x_train[:55000], y_train[:55000], epochs=30, strategy=sg.Strategy.MINI_BATCH,
|
|
70
|
+
batch_size=128, optimizer=sg.Optimizer.ADAMW, learning_rate=1e-3,
|
|
71
|
+
validation_data=(x_train[55000:], y_train[55000:]), monitor=sg.Monitor.VAL_ACCURACY,
|
|
72
|
+
early_stopping_patience=3, restore_best_weights=True)
|
|
73
|
+
print(result.status.name, result.best_epoch, net.evaluate(x_train[55000:], y_train[55000:]))
|
|
74
|
+
|
|
75
|
+
# a convolutional network: images as rows of height x width x channels floats (channels last)
|
|
76
|
+
cnn = sg.Network(sg.Loss.CROSS_ENTROPY, [
|
|
77
|
+
sg.Input(28, 28, 1),
|
|
78
|
+
sg.Conv2D(32, 3, padding=1), sg.MaxPool2D(2), # ReLU and He initialization by default
|
|
79
|
+
sg.Conv2D(64, 3, padding=1), sg.MaxPool2D(2),
|
|
80
|
+
sg.Layer(128, sg.Activation.RELU, sg.Init.HE, dropout=0.3),
|
|
81
|
+
sg.Layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER),
|
|
82
|
+
])
|
|
83
|
+
cnn.train(x_train[:55000], y_train[:55000], epochs=2, strategy=sg.Strategy.MINI_BATCH, batch_size=128,
|
|
84
|
+
optimizer=sg.Optimizer.ADAMW, learning_rate=1e-3)
|
|
85
|
+
print([(l.type.name, l.shape) for l in cnn.layers], cnn.get_weights(0).shape) # filters x 3 x 3 x 1
|
|
86
|
+
|
|
87
|
+
# batch normalization, a depthwise-separable block and image augmentation on CIFAR-10
|
|
88
|
+
x_cifar, y_cifar = sg.load_cifar([f"cifar-10-batches-bin/data_batch_{i}.bin" for i in range(1, 6)])
|
|
89
|
+
bn = sg.Network(sg.Loss.CROSS_ENTROPY, [
|
|
90
|
+
sg.Input(32, 32, 3),
|
|
91
|
+
sg.Conv2D(32, 3, padding=1, activation=sg.Activation.NONE), sg.BatchNorm(sg.Activation.RELU),
|
|
92
|
+
sg.Conv2D(32, 3, padding=1, groups=32, activation=sg.Activation.NONE), sg.BatchNorm(sg.Activation.RELU),
|
|
93
|
+
sg.Conv2D(64, 1, activation=sg.Activation.NONE), sg.BatchNorm(sg.Activation.RELU), sg.MaxPool2D(2),
|
|
94
|
+
sg.Layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER),
|
|
95
|
+
])
|
|
96
|
+
bn.train(x_cifar, y_cifar, epochs=5, strategy=sg.Strategy.MINI_BATCH, batch_size=128,
|
|
97
|
+
optimizer=sg.Optimizer.ADAMW, learning_rate=2e-3, augment_shift=4, augment_flip=True)
|
|
98
|
+
|
|
99
|
+
# a residual block: inputs= names earlier layers, add_add / add_concat combine them
|
|
100
|
+
res = sg.Network(sg.Loss.CROSS_ENTROPY)
|
|
101
|
+
x = res.add_input(32, 32, 3).add_conv2d(16, 3, padding=1, activation=sg.Activation.NONE) \
|
|
102
|
+
.add_batch_norm(sg.Activation.RELU).last
|
|
103
|
+
res.add_conv2d(16, 3, padding=1, activation=sg.Activation.NONE).add_batch_norm(sg.Activation.RELU)
|
|
104
|
+
res.add_conv2d(16, 3, padding=1, activation=sg.Activation.NONE).add_batch_norm()
|
|
105
|
+
res.add_add([x, -1], activation=sg.Activation.RELU) # -1: the layer before
|
|
106
|
+
res.add_global_avg_pool().add_layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER)
|
|
107
|
+
|
|
108
|
+
# PyTorch: a module through ONNX in memory, or a state dict into a network of the same layers
|
|
109
|
+
import torch
|
|
110
|
+
model = torch.nn.Sequential(torch.nn.Conv2d(3, 8, 3, padding=1), torch.nn.ReLU(), torch.nn.Flatten(),
|
|
111
|
+
torch.nn.Linear(8 * 32 * 32, 10))
|
|
112
|
+
net = sg.Network.from_torch(model, torch.rand(1, 3, 32, 32)) # takes (32, 32, 3) channels-last images
|
|
113
|
+
same = sg.Network(sg.Loss.MSE, [sg.Input(32, 32, 3), sg.Conv2D(8, 3, padding=1), sg.Layer(10, sg.Activation.NONE)])
|
|
114
|
+
same.load_pytorch(model.state_dict()) # or "model.pt", "model.safetensors"
|
|
115
|
+
onnx = sg.Network.from_onnx("model.onnx")
|
|
116
|
+
|
|
117
|
+
with sg.Network.load("xor.slett") as net:
|
|
118
|
+
print(net.topology, net.forward([1, 0]))
|
|
119
|
+
# deployment: a read-only INT8 model with integer kernels, and a C header for firmware
|
|
120
|
+
with net.to_model(sg.Precision.INT8) as model:
|
|
121
|
+
print(model.predict([1, 0]), model.layers, model.size)
|
|
122
|
+
net.export_c_header("xor_model.h", "xor_model", sg.Precision.INT8)
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
| API | Notes |
|
|
126
|
+
|---|---|
|
|
127
|
+
| `Network(loss, layers)` / `add_layer(...)` | first layer is the input layer; `layers` may mix `Layer`, `Input`, `Conv2D`, `MaxPool2D`, `AvgPool2D`, `BatchNorm`, `Add`, `Concat`, `GlobalAvgPool` and plain widths |
|
|
128
|
+
| `inputs=` on every `add_*`, `last`, `len(net)` | a layer reads the one before it, or the earlier layers `inputs` names (indices; negative ones count back from the new layer); `last` is the index of the layer added last |
|
|
129
|
+
| `add_add(inputs, activation=NONE, dropout=0)`, `add_concat(inputs, ...)`, `add_global_avg_pool(dropout=0, inputs=None)` | the sum of layers of one shape (residual connections), layers side by side along the channels, the mean of each channel |
|
|
130
|
+
| `Network.from_onnx(path or bytes)`, `Network.from_torch(module, example_input)` | ONNX models (and PyTorch modules, exported to ONNX in memory) as networks with channels-last inputs: transpose NCHW images with `x.transpose(0, 2, 3, 1)` |
|
|
131
|
+
| `load_pytorch(state_dict, path or bytes, modules=None)` | PyTorch weights (`state_dict()`, `torch.save` or safetensors files) into a network of the same layers, reordered for channels-last data; the network is unchanged on error |
|
|
132
|
+
| `add_input(h, w, c)`, `add_conv2d(filters, kernel, stride=1, padding=0, activation=RELU, init=HE, dropout=0, groups=1)`, `add_max_pool2d(kernel, stride=0, padding=0)`, `add_avg_pool2d(...)` | convolution (grouped or depthwise with `groups`) and pooling over channels-last tensors; pooling's stride defaults to the kernel size |
|
|
133
|
+
| `add_batch_norm(activation=NONE, epsilon=1e-5, momentum=0.1, dropout=0)` | per-channel normalization of the previous layer: batch statistics while training, running averages otherwise; `get_running_statistics(i)` / `set_running_statistics(i, mean, var)` |
|
|
134
|
+
| `layers`, `layer(i)`, `topology` | `LayerDescription(type, shape, outputs, activation, dropout, kernel, stride, padding, weight_count, bias_count, groups, epsilon, momentum, inputs)` per layer |
|
|
135
|
+
| `forward(x)` | 1-D input -> vector, 2-D batch -> matrix (one batched `predict()` call; results are copies) |
|
|
136
|
+
| `train(x, y, config=None, validation_data=None, **overrides)` | fields of `TrainConfig` (e.g. `epochs`, `strategy`, `batch_size`, `shuffle`, `early_stopping_patience`, `restore_best_weights`, `augment_shift`, `augment_flip`, `label_smoothing`); returns a `TrainResult`; `callback(network, progress)` gets a `Progress`; exceptions raised in callbacks stop training and are re-raised |
|
|
137
|
+
| `train_from_generator(fn, samples_per_epoch=0, validation_data=None, ...)` | `fn(inputs, targets)` fills the given arrays and returns the number of rows; 0 ends the epoch |
|
|
138
|
+
| `evaluate(x, y)` | `Metrics(loss, accuracy)` |
|
|
139
|
+
| `Trainer(net, max_batch)` | `forward(x)`, `backward(y)`, `backward_output_grads(dl_dout)` for custom losses, `step(optimizer=..., learning_rate=...)`, `zero_grad()`, `train_on_batch(x, y, ...)`; gradients via `net.get_weight_gradients(i)` / `get_bias_gradients(i)`; on the GPU when made with `ComputeMode.VULKAN` |
|
|
140
|
+
| `load_idx(images, labels, num_classes=0)`, `load_cifar(paths, num_classes=10)`, `load_csv(path, target_columns=1, num_classes=0)` | return `(inputs, targets)` float32 arrays; CIFAR images as 32 x 32 x 3, channels last |
|
|
141
|
+
| `save_dataset(path, x, y, input_encoding=AUTO, target_encoding=AUTO, compress=True, shape=None, class_names=None, target_name=None, extra_targets=None)`, `load_dataset(path, target_set=0)`, `dataset_info(path)` | `.slettd` data set files, optionally with the input shape, class names and further sets of targets (`extra_targets`: dicts with `"targets"` and optionally `"name"`, `"class_names"`, `"encoding"`); `dataset_info` returns the counts, encodings, `shape` and `target_sets` |
|
|
142
|
+
| `train_from_file(path, shuffle=True, in_memory=False, prefetch=True, target_set=0, ...)` | trains on a `.slettd` file through the C reader: streamed with a few chunks in memory, or with `in_memory=True` decoded once and kept in its compact form (a byte per 8-bit value); `prefetch=False` keeps decoding off a background thread |
|
|
143
|
+
| `CosineDecay`, `LinearWarmup`, `StepDecay`, `WarmupCosine` | built-in schedules; any `fn(epoch, total, initial_lr)` works too |
|
|
144
|
+
| `get_weights(i)`, `set_weights(i, w)`, `get_biases(i)`, `set_biases(i, b)` | weights `i` feed layer `i + 1`: shape `(out, in)` for a dense layer, `(filters, kernel_h, kernel_w, in_channels / groups)` for a convolution, gamma `(channels,)` for batch normalization (beta are its biases), empty for pooling |
|
|
145
|
+
| `save(path, precision, save_optimizer)`, `Network.load(path)` | `.slett` files, shared with the C API |
|
|
146
|
+
| `to_bytes(precision, save_optimizer=False)`, `Network.from_bytes(data)` | the same files as `bytes` |
|
|
147
|
+
| `to_model(precision=INT8)` | a `Model`: read-only, computes in its precision (integer kernels for INT8, INT4, INT2) |
|
|
148
|
+
| `Model.load(path)`, `Model.from_bytes(data)`, `model.to_bytes()` | models from and to `.slett` files of any version |
|
|
149
|
+
| `model.predict(x)` / `model(x)`, `model.evaluate(x, y)` | 1-D input -> vector, 2-D batch -> matrix; `Metrics(loss, accuracy)` |
|
|
150
|
+
| `model.layers`, `input_size`, `output_size`, `size`, `workspace_size` | `LayerInfo(inputs, outputs, activation, precision, type, shape, input_shape, kernel, stride, padding, groups, epsilon, input_layers)` per layer (normalizations after a dense or convolution layer are folded into it); image and C workspace bytes |
|
|
151
|
+
| `export_c_header(path, name, precision=INT8)` | the model as a C header for the standalone engine (`Spingalett.Inference.h`) |
|
|
152
|
+
| `set_compute_mode`, `set_num_threads`, `seed`, `set_verbose`, `set_log_level`, `set_log_callback` | process-wide settings; `ComputeMode.VULKAN` trains and predicts on the GPU |
|
|
153
|
+
| `set_gpu_precision(Precision.BFLOAT16)`, `get_gpu_precision()` | the GPU's matrix products in bfloat16 on its matrix units, or `FLOAT32` (the default); returns whether the GPU multiplies in that precision |
|
|
154
|
+
| `cpu_kernels()`, `gpu_device()`, `library_version()`, `library_path()` | the matrix kernels in use (`"AVX-512"`, `"AVX2"`, ...), the GPU `VULKAN` uses (`None` without one), the loaded library |
|
|
155
|
+
|
|
156
|
+
`uint8` arrays stand for 8-bit images: value q means q / 255 in `train()`, `forward()`,
|
|
157
|
+
`predict()`, `evaluate()` and `save_dataset()`. `train()` keeps such inputs as bytes, a quarter of
|
|
158
|
+
the memory of float32, and converts them a batch at a time, with the same result as training on
|
|
159
|
+
`x / 255` in float32.
|
|
160
|
+
|
|
161
|
+
Library errors raise `SpingalettError`, whose `code` is an `ErrorCode`. The bindings check on import
|
|
162
|
+
that the library has the same major.minor version (`library_version()`), because they mirror its
|
|
163
|
+
struct layouts. A `Network` is not thread-safe: use one per thread. A `Model` is read-only and can
|
|
164
|
+
be shared between threads.
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# Spingalett for Python
|
|
2
|
+
|
|
3
|
+
Thin `ctypes` bindings over the Spingalett shared library: no compiler is needed to install them,
|
|
4
|
+
only NumPy and `libspingalett` (`.so` / `.dylib` / `.dll`). The wheels on PyPI and on the GitHub
|
|
5
|
+
releases carry the library inside (Linux x86-64 and AArch64 with glibc 2.28 or newer, Windows
|
|
6
|
+
x86-64, macOS 11 and newer, for any Python 3), with OpenMP:
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
pip install spingalett
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
From a source checkout, build the library and point the bindings at it:
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
cmake -S . -B Build -DCMAKE_BUILD_TYPE=Release -DBUILD_WITH_OPENMP=ON
|
|
16
|
+
cmake --build Build --parallel # produces Bin/libspingalett.so
|
|
17
|
+
pip install ./Bindings/Python
|
|
18
|
+
export SPINGALETT_LIBRARY=$PWD/Bin/libspingalett.so # or put Bin/ on the library path
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
Without `SPINGALETT_LIBRARY` the package looks inside itself (wheels), then in the repository's
|
|
22
|
+
`Bin/` directory (when imported from a checkout), then on the system library path. The package is
|
|
23
|
+
typed (`py.typed`): editors and type checkers read its annotations.
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
import numpy as np
|
|
27
|
+
import spingalett as sg
|
|
28
|
+
|
|
29
|
+
x = np.array([[0, 0], [0, 1], [1, 0], [1, 1]], dtype=np.float32)
|
|
30
|
+
y = np.array([[0], [1], [1], [0]], dtype=np.float32)
|
|
31
|
+
|
|
32
|
+
sg.set_compute_mode(sg.ComputeMode.OPENBLAS) # falls back to single-threaded if unavailable
|
|
33
|
+
sg.seed(42)
|
|
34
|
+
|
|
35
|
+
layers = [
|
|
36
|
+
sg.Layer(2),
|
|
37
|
+
sg.Layer(16, sg.Activation.TANH, sg.Init.XAVIER, dropout=0.1),
|
|
38
|
+
sg.Layer(1, sg.Activation.SIGMOID, sg.Init.XAVIER),
|
|
39
|
+
]
|
|
40
|
+
with sg.Network(sg.Loss.MSE, layers) as net:
|
|
41
|
+
net.train(x, y, epochs=3000, optimizer=sg.Optimizer.ADAMW, learning_rate=0.02,
|
|
42
|
+
weight_decay=1e-4, lr_scheduler=sg.WarmupCosine(warmup_epochs=100),
|
|
43
|
+
callback=lambda net, progress: progress.train_loss < 1e-4, callback_interval=100)
|
|
44
|
+
print(net.forward(x)) # one row per sample
|
|
45
|
+
net.save("xor", precision=sg.Precision.FP16)
|
|
46
|
+
|
|
47
|
+
# validation, early stopping and the best epoch's weights
|
|
48
|
+
x_train, y_train = sg.load_idx("train-images-idx3-ubyte", "train-labels-idx1-ubyte", num_classes=10)
|
|
49
|
+
with sg.Network(sg.Loss.CROSS_ENTROPY, [784, sg.Layer(128, sg.Activation.RELU, sg.Init.HE),
|
|
50
|
+
sg.Layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER)]) as net:
|
|
51
|
+
result = net.train(x_train[:55000], y_train[:55000], epochs=30, strategy=sg.Strategy.MINI_BATCH,
|
|
52
|
+
batch_size=128, optimizer=sg.Optimizer.ADAMW, learning_rate=1e-3,
|
|
53
|
+
validation_data=(x_train[55000:], y_train[55000:]), monitor=sg.Monitor.VAL_ACCURACY,
|
|
54
|
+
early_stopping_patience=3, restore_best_weights=True)
|
|
55
|
+
print(result.status.name, result.best_epoch, net.evaluate(x_train[55000:], y_train[55000:]))
|
|
56
|
+
|
|
57
|
+
# a convolutional network: images as rows of height x width x channels floats (channels last)
|
|
58
|
+
cnn = sg.Network(sg.Loss.CROSS_ENTROPY, [
|
|
59
|
+
sg.Input(28, 28, 1),
|
|
60
|
+
sg.Conv2D(32, 3, padding=1), sg.MaxPool2D(2), # ReLU and He initialization by default
|
|
61
|
+
sg.Conv2D(64, 3, padding=1), sg.MaxPool2D(2),
|
|
62
|
+
sg.Layer(128, sg.Activation.RELU, sg.Init.HE, dropout=0.3),
|
|
63
|
+
sg.Layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER),
|
|
64
|
+
])
|
|
65
|
+
cnn.train(x_train[:55000], y_train[:55000], epochs=2, strategy=sg.Strategy.MINI_BATCH, batch_size=128,
|
|
66
|
+
optimizer=sg.Optimizer.ADAMW, learning_rate=1e-3)
|
|
67
|
+
print([(l.type.name, l.shape) for l in cnn.layers], cnn.get_weights(0).shape) # filters x 3 x 3 x 1
|
|
68
|
+
|
|
69
|
+
# batch normalization, a depthwise-separable block and image augmentation on CIFAR-10
|
|
70
|
+
x_cifar, y_cifar = sg.load_cifar([f"cifar-10-batches-bin/data_batch_{i}.bin" for i in range(1, 6)])
|
|
71
|
+
bn = sg.Network(sg.Loss.CROSS_ENTROPY, [
|
|
72
|
+
sg.Input(32, 32, 3),
|
|
73
|
+
sg.Conv2D(32, 3, padding=1, activation=sg.Activation.NONE), sg.BatchNorm(sg.Activation.RELU),
|
|
74
|
+
sg.Conv2D(32, 3, padding=1, groups=32, activation=sg.Activation.NONE), sg.BatchNorm(sg.Activation.RELU),
|
|
75
|
+
sg.Conv2D(64, 1, activation=sg.Activation.NONE), sg.BatchNorm(sg.Activation.RELU), sg.MaxPool2D(2),
|
|
76
|
+
sg.Layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER),
|
|
77
|
+
])
|
|
78
|
+
bn.train(x_cifar, y_cifar, epochs=5, strategy=sg.Strategy.MINI_BATCH, batch_size=128,
|
|
79
|
+
optimizer=sg.Optimizer.ADAMW, learning_rate=2e-3, augment_shift=4, augment_flip=True)
|
|
80
|
+
|
|
81
|
+
# a residual block: inputs= names earlier layers, add_add / add_concat combine them
|
|
82
|
+
res = sg.Network(sg.Loss.CROSS_ENTROPY)
|
|
83
|
+
x = res.add_input(32, 32, 3).add_conv2d(16, 3, padding=1, activation=sg.Activation.NONE) \
|
|
84
|
+
.add_batch_norm(sg.Activation.RELU).last
|
|
85
|
+
res.add_conv2d(16, 3, padding=1, activation=sg.Activation.NONE).add_batch_norm(sg.Activation.RELU)
|
|
86
|
+
res.add_conv2d(16, 3, padding=1, activation=sg.Activation.NONE).add_batch_norm()
|
|
87
|
+
res.add_add([x, -1], activation=sg.Activation.RELU) # -1: the layer before
|
|
88
|
+
res.add_global_avg_pool().add_layer(10, sg.Activation.SOFTMAX, sg.Init.XAVIER)
|
|
89
|
+
|
|
90
|
+
# PyTorch: a module through ONNX in memory, or a state dict into a network of the same layers
|
|
91
|
+
import torch
|
|
92
|
+
model = torch.nn.Sequential(torch.nn.Conv2d(3, 8, 3, padding=1), torch.nn.ReLU(), torch.nn.Flatten(),
|
|
93
|
+
torch.nn.Linear(8 * 32 * 32, 10))
|
|
94
|
+
net = sg.Network.from_torch(model, torch.rand(1, 3, 32, 32)) # takes (32, 32, 3) channels-last images
|
|
95
|
+
same = sg.Network(sg.Loss.MSE, [sg.Input(32, 32, 3), sg.Conv2D(8, 3, padding=1), sg.Layer(10, sg.Activation.NONE)])
|
|
96
|
+
same.load_pytorch(model.state_dict()) # or "model.pt", "model.safetensors"
|
|
97
|
+
onnx = sg.Network.from_onnx("model.onnx")
|
|
98
|
+
|
|
99
|
+
with sg.Network.load("xor.slett") as net:
|
|
100
|
+
print(net.topology, net.forward([1, 0]))
|
|
101
|
+
# deployment: a read-only INT8 model with integer kernels, and a C header for firmware
|
|
102
|
+
with net.to_model(sg.Precision.INT8) as model:
|
|
103
|
+
print(model.predict([1, 0]), model.layers, model.size)
|
|
104
|
+
net.export_c_header("xor_model.h", "xor_model", sg.Precision.INT8)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
| API | Notes |
|
|
108
|
+
|---|---|
|
|
109
|
+
| `Network(loss, layers)` / `add_layer(...)` | first layer is the input layer; `layers` may mix `Layer`, `Input`, `Conv2D`, `MaxPool2D`, `AvgPool2D`, `BatchNorm`, `Add`, `Concat`, `GlobalAvgPool` and plain widths |
|
|
110
|
+
| `inputs=` on every `add_*`, `last`, `len(net)` | a layer reads the one before it, or the earlier layers `inputs` names (indices; negative ones count back from the new layer); `last` is the index of the layer added last |
|
|
111
|
+
| `add_add(inputs, activation=NONE, dropout=0)`, `add_concat(inputs, ...)`, `add_global_avg_pool(dropout=0, inputs=None)` | the sum of layers of one shape (residual connections), layers side by side along the channels, the mean of each channel |
|
|
112
|
+
| `Network.from_onnx(path or bytes)`, `Network.from_torch(module, example_input)` | ONNX models (and PyTorch modules, exported to ONNX in memory) as networks with channels-last inputs: transpose NCHW images with `x.transpose(0, 2, 3, 1)` |
|
|
113
|
+
| `load_pytorch(state_dict, path or bytes, modules=None)` | PyTorch weights (`state_dict()`, `torch.save` or safetensors files) into a network of the same layers, reordered for channels-last data; the network is unchanged on error |
|
|
114
|
+
| `add_input(h, w, c)`, `add_conv2d(filters, kernel, stride=1, padding=0, activation=RELU, init=HE, dropout=0, groups=1)`, `add_max_pool2d(kernel, stride=0, padding=0)`, `add_avg_pool2d(...)` | convolution (grouped or depthwise with `groups`) and pooling over channels-last tensors; pooling's stride defaults to the kernel size |
|
|
115
|
+
| `add_batch_norm(activation=NONE, epsilon=1e-5, momentum=0.1, dropout=0)` | per-channel normalization of the previous layer: batch statistics while training, running averages otherwise; `get_running_statistics(i)` / `set_running_statistics(i, mean, var)` |
|
|
116
|
+
| `layers`, `layer(i)`, `topology` | `LayerDescription(type, shape, outputs, activation, dropout, kernel, stride, padding, weight_count, bias_count, groups, epsilon, momentum, inputs)` per layer |
|
|
117
|
+
| `forward(x)` | 1-D input -> vector, 2-D batch -> matrix (one batched `predict()` call; results are copies) |
|
|
118
|
+
| `train(x, y, config=None, validation_data=None, **overrides)` | fields of `TrainConfig` (e.g. `epochs`, `strategy`, `batch_size`, `shuffle`, `early_stopping_patience`, `restore_best_weights`, `augment_shift`, `augment_flip`, `label_smoothing`); returns a `TrainResult`; `callback(network, progress)` gets a `Progress`; exceptions raised in callbacks stop training and are re-raised |
|
|
119
|
+
| `train_from_generator(fn, samples_per_epoch=0, validation_data=None, ...)` | `fn(inputs, targets)` fills the given arrays and returns the number of rows; 0 ends the epoch |
|
|
120
|
+
| `evaluate(x, y)` | `Metrics(loss, accuracy)` |
|
|
121
|
+
| `Trainer(net, max_batch)` | `forward(x)`, `backward(y)`, `backward_output_grads(dl_dout)` for custom losses, `step(optimizer=..., learning_rate=...)`, `zero_grad()`, `train_on_batch(x, y, ...)`; gradients via `net.get_weight_gradients(i)` / `get_bias_gradients(i)`; on the GPU when made with `ComputeMode.VULKAN` |
|
|
122
|
+
| `load_idx(images, labels, num_classes=0)`, `load_cifar(paths, num_classes=10)`, `load_csv(path, target_columns=1, num_classes=0)` | return `(inputs, targets)` float32 arrays; CIFAR images as 32 x 32 x 3, channels last |
|
|
123
|
+
| `save_dataset(path, x, y, input_encoding=AUTO, target_encoding=AUTO, compress=True, shape=None, class_names=None, target_name=None, extra_targets=None)`, `load_dataset(path, target_set=0)`, `dataset_info(path)` | `.slettd` data set files, optionally with the input shape, class names and further sets of targets (`extra_targets`: dicts with `"targets"` and optionally `"name"`, `"class_names"`, `"encoding"`); `dataset_info` returns the counts, encodings, `shape` and `target_sets` |
|
|
124
|
+
| `train_from_file(path, shuffle=True, in_memory=False, prefetch=True, target_set=0, ...)` | trains on a `.slettd` file through the C reader: streamed with a few chunks in memory, or with `in_memory=True` decoded once and kept in its compact form (a byte per 8-bit value); `prefetch=False` keeps decoding off a background thread |
|
|
125
|
+
| `CosineDecay`, `LinearWarmup`, `StepDecay`, `WarmupCosine` | built-in schedules; any `fn(epoch, total, initial_lr)` works too |
|
|
126
|
+
| `get_weights(i)`, `set_weights(i, w)`, `get_biases(i)`, `set_biases(i, b)` | weights `i` feed layer `i + 1`: shape `(out, in)` for a dense layer, `(filters, kernel_h, kernel_w, in_channels / groups)` for a convolution, gamma `(channels,)` for batch normalization (beta are its biases), empty for pooling |
|
|
127
|
+
| `save(path, precision, save_optimizer)`, `Network.load(path)` | `.slett` files, shared with the C API |
|
|
128
|
+
| `to_bytes(precision, save_optimizer=False)`, `Network.from_bytes(data)` | the same files as `bytes` |
|
|
129
|
+
| `to_model(precision=INT8)` | a `Model`: read-only, computes in its precision (integer kernels for INT8, INT4, INT2) |
|
|
130
|
+
| `Model.load(path)`, `Model.from_bytes(data)`, `model.to_bytes()` | models from and to `.slett` files of any version |
|
|
131
|
+
| `model.predict(x)` / `model(x)`, `model.evaluate(x, y)` | 1-D input -> vector, 2-D batch -> matrix; `Metrics(loss, accuracy)` |
|
|
132
|
+
| `model.layers`, `input_size`, `output_size`, `size`, `workspace_size` | `LayerInfo(inputs, outputs, activation, precision, type, shape, input_shape, kernel, stride, padding, groups, epsilon, input_layers)` per layer (normalizations after a dense or convolution layer are folded into it); image and C workspace bytes |
|
|
133
|
+
| `export_c_header(path, name, precision=INT8)` | the model as a C header for the standalone engine (`Spingalett.Inference.h`) |
|
|
134
|
+
| `set_compute_mode`, `set_num_threads`, `seed`, `set_verbose`, `set_log_level`, `set_log_callback` | process-wide settings; `ComputeMode.VULKAN` trains and predicts on the GPU |
|
|
135
|
+
| `set_gpu_precision(Precision.BFLOAT16)`, `get_gpu_precision()` | the GPU's matrix products in bfloat16 on its matrix units, or `FLOAT32` (the default); returns whether the GPU multiplies in that precision |
|
|
136
|
+
| `cpu_kernels()`, `gpu_device()`, `library_version()`, `library_path()` | the matrix kernels in use (`"AVX-512"`, `"AVX2"`, ...), the GPU `VULKAN` uses (`None` without one), the loaded library |
|
|
137
|
+
|
|
138
|
+
`uint8` arrays stand for 8-bit images: value q means q / 255 in `train()`, `forward()`,
|
|
139
|
+
`predict()`, `evaluate()` and `save_dataset()`. `train()` keeps such inputs as bytes, a quarter of
|
|
140
|
+
the memory of float32, and converts them a batch at a time, with the same result as training on
|
|
141
|
+
`x / 255` in float32.
|
|
142
|
+
|
|
143
|
+
Library errors raise `SpingalettError`, whose `code` is an `ErrorCode`. The bindings check on import
|
|
144
|
+
that the library has the same major.minor version (`library_version()`), because they mirror its
|
|
145
|
+
struct layouts. A `Network` is not thread-safe: use one per thread. A `Model` is read-only and can
|
|
146
|
+
be shared between threads.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "spingalett"
|
|
7
|
+
version = "0.12.0"
|
|
8
|
+
description = "Neural networks in C23 from Python: training, graphs, ONNX and PyTorch import, INT8 to INT2 deployment"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
dependencies = ["numpy>=1.20"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Programming Language :: Python :: 3",
|
|
15
|
+
"Programming Language :: C",
|
|
16
|
+
"Operating System :: POSIX :: Linux",
|
|
17
|
+
"Operating System :: MacOS",
|
|
18
|
+
"Operating System :: Microsoft :: Windows",
|
|
19
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
20
|
+
"Typing :: Typed",
|
|
21
|
+
]
|
|
22
|
+
keywords = ["neural networks", "deep learning", "inference", "quantization", "onnx", "microcontrollers"]
|
|
23
|
+
|
|
24
|
+
[project.urls]
|
|
25
|
+
Homepage = "https://github.com/pka-human/Spingalett"
|
|
26
|
+
|
|
27
|
+
[tool.setuptools]
|
|
28
|
+
packages = ["spingalett"]
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
"""Builds the spingalett wheel. The package is pure Python over ctypes; when a shared library has been
|
|
3
|
+
copied into spingalett/ (libspingalett.so.*, libspingalett.*.dylib or libspingalett.dll, with the
|
|
4
|
+
runtime libraries it loads), the wheel carries it and is tagged for the platform but for any
|
|
5
|
+
Python 3 (py3-none-<platform>), since nothing in it is built against Python. Without a library the
|
|
6
|
+
wheel is pure and finds the library at run time (see spingalett/__init__.py)."""
|
|
7
|
+
import glob
|
|
8
|
+
import os
|
|
9
|
+
|
|
10
|
+
from setuptools import setup
|
|
11
|
+
from setuptools.dist import Distribution
|
|
12
|
+
|
|
13
|
+
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
14
|
+
LIBRARIES = [os.path.basename(p) for pattern in ("*.so*", "*.dylib", "*.dll")
|
|
15
|
+
for p in glob.glob(os.path.join(HERE, "spingalett", pattern))]
|
|
16
|
+
|
|
17
|
+
try:
|
|
18
|
+
from setuptools.command.bdist_wheel import bdist_wheel
|
|
19
|
+
except ImportError: # setuptools before 70.1
|
|
20
|
+
from wheel.bdist_wheel import bdist_wheel
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class PlatformWheel(bdist_wheel):
|
|
24
|
+
def finalize_options(self):
|
|
25
|
+
super().finalize_options()
|
|
26
|
+
self.root_is_pure = not LIBRARIES
|
|
27
|
+
|
|
28
|
+
def get_tag(self):
|
|
29
|
+
python, abi, platform = super().get_tag()
|
|
30
|
+
return ("py3", "none", platform) if LIBRARIES else (python, abi, platform)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class BinaryDistribution(Distribution):
|
|
34
|
+
def has_ext_modules(self):
|
|
35
|
+
return bool(LIBRARIES)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
setup(distclass=BinaryDistribution, cmdclass={"bdist_wheel": PlatformWheel},
|
|
39
|
+
package_data={"spingalett": ["py.typed"] + LIBRARIES})
|