forge-dl 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- forge/__init__.py +19 -0
- forge/autograd/__init__.py +20 -0
- forge/autograd/accelerate_backend.py +70 -0
- forge/autograd/engine.py +14 -0
- forge/autograd/fused_operations.py +120 -0
- forge/autograd/fusion.py +40 -0
- forge/autograd/grad_check.py +58 -0
- forge/autograd/mps_backend.py +52 -0
- forge/autograd/operations.py +1063 -0
- forge/compiler/__init__.py +23 -0
- forge/compiler/build.py +36 -0
- forge/compiler/codegen.py +99 -0
- forge/compiler/compiled_run.py +127 -0
- forge/compiler/fusion.py +62 -0
- forge/compiler/graph.py +108 -0
- forge/compiler/interpreter.py +109 -0
- forge/compiler/kernels.c +195 -0
- forge/dtype.py +26 -0
- forge/nn/__init__.py +23 -0
- forge/nn/layers.py +444 -0
- forge/nn/losses.py +122 -0
- forge/nn/module.py +92 -0
- forge/nn/parameter.py +13 -0
- forge/optim/__init__.py +8 -0
- forge/optim/optimizer.py +85 -0
- forge/serialization.py +60 -0
- forge/tensor.py +474 -0
- forge_dl-0.1.0.dist-info/METADATA +207 -0
- forge_dl-0.1.0.dist-info/RECORD +32 -0
- forge_dl-0.1.0.dist-info/WHEEL +5 -0
- forge_dl-0.1.0.dist-info/licenses/LICENSE +21 -0
- forge_dl-0.1.0.dist-info/top_level.txt +1 -0
forge/__init__.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Forge: a machine learning library and deep learning compiler written from
|
|
3
|
+
scratch in Python.
|
|
4
|
+
|
|
5
|
+
The core has no dependencies. Every operation, from the autograd engine to the
|
|
6
|
+
attention mechanism, is implemented from first principles.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from forge.tensor import Tensor
|
|
10
|
+
from forge.dtype import float32, float64, int32, int64
|
|
11
|
+
from forge.serialization import save_model, load_model
|
|
12
|
+
|
|
13
|
+
__version__ = "0.1.0"
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"Tensor",
|
|
17
|
+
"float32", "float64", "int32", "int64",
|
|
18
|
+
"save_model", "load_model",
|
|
19
|
+
]
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The autograd engine: reverse-mode automatic differentiation on a define-by-run
|
|
3
|
+
graph.
|
|
4
|
+
|
|
5
|
+
Every operation records itself as it executes, and backward() walks that record
|
|
6
|
+
in reverse, applying the chain rule at each node.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from forge.autograd.engine import Function
|
|
10
|
+
from forge.autograd.operations import (
|
|
11
|
+
Add, Mul, Sub, Pow, Neg, Sum, Mean, Matmul,
|
|
12
|
+
ReLU, Sigmoid, Tanh, Log, Clamp, Softmax, Exp,
|
|
13
|
+
)
|
|
14
|
+
from forge.autograd.grad_check import grad_check
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"Function", "grad_check",
|
|
18
|
+
"Add", "Mul", "Sub", "Pow", "Neg", "Sum", "Mean", "Matmul",
|
|
19
|
+
"ReLU", "Sigmoid", "Tanh", "Log", "Clamp", "Softmax", "Exp",
|
|
20
|
+
]
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import array
|
|
2
|
+
import ctypes
|
|
3
|
+
|
|
4
|
+
try:
|
|
5
|
+
_lib = ctypes.CDLL("/System/Library/Frameworks/Accelerate.framework/Accelerate")
|
|
6
|
+
|
|
7
|
+
cblas_sgemm = _lib.cblas_sgemm
|
|
8
|
+
|
|
9
|
+
float_pointer = ctypes.POINTER(ctypes.c_float)
|
|
10
|
+
|
|
11
|
+
cblas_sgemm.argtypes = [
|
|
12
|
+
ctypes.c_int, # storage order (row-major vs column-major)
|
|
13
|
+
ctypes.c_int, # whether to transpose the left matrix
|
|
14
|
+
ctypes.c_int, # whether to transpose the right matrix
|
|
15
|
+
ctypes.c_int, # number of rows in the result
|
|
16
|
+
ctypes.c_int, # number of columns in the result
|
|
17
|
+
ctypes.c_int, # the shared inner dimension being summed over
|
|
18
|
+
ctypes.c_float, # alpha: the product is scaled by this
|
|
19
|
+
float_pointer, # pointer to the left matrix's data
|
|
20
|
+
ctypes.c_int, # row stride of the left matrix
|
|
21
|
+
float_pointer, # pointer to the right matrix's data
|
|
22
|
+
ctypes.c_int, # row stride of the right matrix
|
|
23
|
+
ctypes.c_float, # beta: the existing result is scaled by this
|
|
24
|
+
float_pointer, # pointer to the output matrix's data
|
|
25
|
+
ctypes.c_int, # row stride of the output matrix
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
cblas_sgemm.restype = None
|
|
29
|
+
|
|
30
|
+
ACCELERATE_AVAILABLE = True
|
|
31
|
+
|
|
32
|
+
# Only occurs on non-macOS systems
|
|
33
|
+
except Exception:
|
|
34
|
+
ACCELERATE_AVAILABLE = False
|
|
35
|
+
|
|
36
|
+
CBLAS_ROW_MAJOR = 101
|
|
37
|
+
CBLAS_NO_TRANSPOSE = 111
|
|
38
|
+
|
|
39
|
+
BYTES_PER_FLOAT = 4
|
|
40
|
+
|
|
41
|
+
def accelerate_matmul(left_data, right_data, left_rows, shared_dim, right_cols):
|
|
42
|
+
"""
|
|
43
|
+
Multiplies two matrices on the CPU using Accelerate.
|
|
44
|
+
|
|
45
|
+
Returns a flat float32 array of the result.
|
|
46
|
+
"""
|
|
47
|
+
result_data = array.array('f', bytes(BYTES_PER_FLOAT * left_rows * right_cols))
|
|
48
|
+
|
|
49
|
+
left_pointer = ctypes.cast(left_data.buffer_info()[0], float_pointer)
|
|
50
|
+
right_pointer = ctypes.cast(right_data.buffer_info()[0], float_pointer)
|
|
51
|
+
result_pointer = ctypes.cast(result_data.buffer_info()[0], float_pointer)
|
|
52
|
+
|
|
53
|
+
cblas_sgemm(
|
|
54
|
+
CBLAS_ROW_MAJOR, # data is laid out row by row
|
|
55
|
+
CBLAS_NO_TRANSPOSE, # do not transpose the left matrix
|
|
56
|
+
CBLAS_NO_TRANSPOSE, # do not transpose the right matrix
|
|
57
|
+
left_rows, # rows of the result
|
|
58
|
+
right_cols, # columns of the result
|
|
59
|
+
shared_dim, # the inner dimension summed over
|
|
60
|
+
1.0, # alpha: scale the product by 1.0
|
|
61
|
+
left_pointer, # the left matrix
|
|
62
|
+
shared_dim, # left's row stride
|
|
63
|
+
right_pointer, # the right matrix
|
|
64
|
+
right_cols, # right's row stride
|
|
65
|
+
0.0, # beta: ignore the zeroed existing result
|
|
66
|
+
result_pointer, # where to write the answer
|
|
67
|
+
right_cols, # result's row stride
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
return result_data
|
forge/autograd/engine.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
class Function:
|
|
2
|
+
"""Base class for all differentiable operations"""
|
|
3
|
+
def __init__(self):
|
|
4
|
+
self.inputs = []
|
|
5
|
+
self.saved_tensors = []
|
|
6
|
+
|
|
7
|
+
def save_for_backward(self, *tensors):
|
|
8
|
+
self.saved_tensors = list(tensors)
|
|
9
|
+
|
|
10
|
+
def forward(self, *args):
|
|
11
|
+
raise NotImplementedError
|
|
12
|
+
|
|
13
|
+
def backward(self, grad_output):
|
|
14
|
+
raise NotImplementedError
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
import array as _array
|
|
2
|
+
import math
|
|
3
|
+
from forge.autograd.engine import Function
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class FusedLinearReLU(Function):
|
|
7
|
+
"""Fused matmul + bias add + ReLU in a single pass"""
|
|
8
|
+
|
|
9
|
+
def forward(self, x, weight_t, bias):
|
|
10
|
+
self.inputs = [x, weight_t, bias]
|
|
11
|
+
self.save_for_backward(x, weight_t, bias)
|
|
12
|
+
from forge.tensor import Tensor
|
|
13
|
+
|
|
14
|
+
m = x.shape[0]
|
|
15
|
+
n = x.shape[1]
|
|
16
|
+
p = weight_t.shape[1]
|
|
17
|
+
|
|
18
|
+
new_data = _array.array(x.dtype.typecode, [])
|
|
19
|
+
|
|
20
|
+
for i in range(m):
|
|
21
|
+
for j in range(p):
|
|
22
|
+
# Matmul
|
|
23
|
+
total = 0.0
|
|
24
|
+
for k in range(n):
|
|
25
|
+
total += x._data[i * n + k] * weight_t._data[k * p + j]
|
|
26
|
+
|
|
27
|
+
# Add bias
|
|
28
|
+
total += bias._data[j]
|
|
29
|
+
|
|
30
|
+
# ReLU
|
|
31
|
+
if total < 0:
|
|
32
|
+
total = 0.0
|
|
33
|
+
|
|
34
|
+
new_data.append(total)
|
|
35
|
+
|
|
36
|
+
result = Tensor.__new__(Tensor)
|
|
37
|
+
result._data = new_data
|
|
38
|
+
result.shape = (m, p)
|
|
39
|
+
result.dtype = x.dtype
|
|
40
|
+
result.requires_grad = False
|
|
41
|
+
result.grad = None
|
|
42
|
+
result._grad_fn = None
|
|
43
|
+
|
|
44
|
+
# Save output for backward (need to know where ReLU killed values)
|
|
45
|
+
self._output = result
|
|
46
|
+
return result
|
|
47
|
+
|
|
48
|
+
def backward(self, grad_output):
|
|
49
|
+
x, weight_t, bias = self.saved_tensors
|
|
50
|
+
output = self._output
|
|
51
|
+
from forge.tensor import Tensor, _broadcast_shape, _broadcast_data
|
|
52
|
+
from forge.autograd.operations import _unbroadcast
|
|
53
|
+
|
|
54
|
+
m = x.shape[0]
|
|
55
|
+
n = x.shape[1]
|
|
56
|
+
p = weight_t.shape[1]
|
|
57
|
+
|
|
58
|
+
# Apply ReLU mask to grad_output
|
|
59
|
+
relu_grad_data = _array.array(
|
|
60
|
+
grad_output.dtype.typecode,
|
|
61
|
+
[g if o > 0 else 0.0 for g, o in zip(grad_output._data, output._data)]
|
|
62
|
+
)
|
|
63
|
+
relu_grad = Tensor.__new__(Tensor)
|
|
64
|
+
relu_grad._data = relu_grad_data
|
|
65
|
+
relu_grad.shape = grad_output.shape
|
|
66
|
+
relu_grad.dtype = grad_output.dtype
|
|
67
|
+
relu_grad.requires_grad = False
|
|
68
|
+
relu_grad.grad = None
|
|
69
|
+
relu_grad._grad_fn = None
|
|
70
|
+
|
|
71
|
+
# Grad for x: relu_grad @ weight_t.T
|
|
72
|
+
# weight_t is (n, p), weight_t.T is (p, n)
|
|
73
|
+
grad_x_data = _array.array(x.dtype.typecode, [0.0] * (m * n))
|
|
74
|
+
for i in range(m):
|
|
75
|
+
for j in range(n):
|
|
76
|
+
total = 0.0
|
|
77
|
+
for k in range(p):
|
|
78
|
+
total += relu_grad._data[i * p + k] * weight_t._data[j * p + k]
|
|
79
|
+
grad_x_data[i * n + j] = total
|
|
80
|
+
|
|
81
|
+
grad_x = Tensor.__new__(Tensor)
|
|
82
|
+
grad_x._data = grad_x_data
|
|
83
|
+
grad_x.shape = x.shape
|
|
84
|
+
grad_x.dtype = x.dtype
|
|
85
|
+
grad_x.requires_grad = False
|
|
86
|
+
grad_x.grad = None
|
|
87
|
+
grad_x._grad_fn = None
|
|
88
|
+
|
|
89
|
+
# Grad for weight_t: x.T @ relu_grad
|
|
90
|
+
grad_wt_data = _array.array(x.dtype.typecode, [0.0] * (n * p))
|
|
91
|
+
for i in range(n):
|
|
92
|
+
for j in range(p):
|
|
93
|
+
total = 0.0
|
|
94
|
+
for k in range(m):
|
|
95
|
+
total += x._data[k * n + i] * relu_grad._data[k * p + j]
|
|
96
|
+
grad_wt_data[i * p + j] = total
|
|
97
|
+
|
|
98
|
+
grad_wt = Tensor.__new__(Tensor)
|
|
99
|
+
grad_wt._data = grad_wt_data
|
|
100
|
+
grad_wt.shape = weight_t.shape
|
|
101
|
+
grad_wt.dtype = weight_t.dtype
|
|
102
|
+
grad_wt.requires_grad = False
|
|
103
|
+
grad_wt.grad = None
|
|
104
|
+
grad_wt._grad_fn = None
|
|
105
|
+
|
|
106
|
+
# Grad for bias: sum relu_grad along rows
|
|
107
|
+
grad_bias_data = _array.array(bias.dtype.typecode, [0.0] * p)
|
|
108
|
+
for i in range(m):
|
|
109
|
+
for j in range(p):
|
|
110
|
+
grad_bias_data[j] += relu_grad._data[i * p + j]
|
|
111
|
+
|
|
112
|
+
grad_bias = Tensor.__new__(Tensor)
|
|
113
|
+
grad_bias._data = grad_bias_data
|
|
114
|
+
grad_bias.shape = bias.shape
|
|
115
|
+
grad_bias.dtype = bias.dtype
|
|
116
|
+
grad_bias.requires_grad = False
|
|
117
|
+
grad_bias.grad = None
|
|
118
|
+
grad_bias._grad_fn = None
|
|
119
|
+
|
|
120
|
+
return grad_x, grad_wt, grad_bias
|
forge/autograd/fusion.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
from forge.nn.layers import Linear, ReLULayer, FusedLinearReLULayer
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def optimize_model(layers):
|
|
5
|
+
"""
|
|
6
|
+
Takes a list of layers and returns an optimized list
|
|
7
|
+
with fused operations where possible.
|
|
8
|
+
"""
|
|
9
|
+
optimized = []
|
|
10
|
+
i = 0
|
|
11
|
+
|
|
12
|
+
while i < len(layers):
|
|
13
|
+
# Pattern: Linear followed by ReLU
|
|
14
|
+
if (i + 1 < len(layers) and isinstance(layers[i], Linear) and isinstance(layers[i + 1], ReLULayer)):
|
|
15
|
+
linear = layers[i]
|
|
16
|
+
in_f = linear.weight.shape[1]
|
|
17
|
+
out_f = linear.weight.shape[0]
|
|
18
|
+
|
|
19
|
+
fused = FusedLinearReLULayer(in_f, out_f)
|
|
20
|
+
|
|
21
|
+
# Copy weights from the original linear layer
|
|
22
|
+
fused.weight._data = linear.weight._data.__class__(
|
|
23
|
+
linear.weight.dtype.typecode, linear.weight._data
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
# Copy bias — need to flatten from (1, out_f) to (out_f,)
|
|
27
|
+
if linear.bias is not None:
|
|
28
|
+
import array as _array
|
|
29
|
+
fused.bias._data = _array.array(
|
|
30
|
+
linear.bias.dtype.typecode, list(linear.bias._data)
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
optimized.append(fused)
|
|
34
|
+
i += 2 # Skip both the Linear and ReLU
|
|
35
|
+
|
|
36
|
+
else:
|
|
37
|
+
optimized.append(layers[i])
|
|
38
|
+
i += 1
|
|
39
|
+
|
|
40
|
+
return optimized
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""
|
|
2
|
+
It is important to draw wisdom from many different places.
|
|
3
|
+
If you take it from only one place, it becomes rigid and stale.
|
|
4
|
+
- Uncle Iroh
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
def grad_check(func, inputs, eps=1e-3, tol=1e-2):
|
|
8
|
+
"""
|
|
9
|
+
Using first principles for differentiation: df/dx = lim(h->0) [f(x + h) - f(x - h)] / 2h
|
|
10
|
+
We use a small value of h to approximate the limit.
|
|
11
|
+
"""
|
|
12
|
+
for inp in inputs:
|
|
13
|
+
inp.grad = None
|
|
14
|
+
|
|
15
|
+
output = func(*inputs)
|
|
16
|
+
output.backward()
|
|
17
|
+
|
|
18
|
+
analytical_grads = [inp.grad for inp in inputs]
|
|
19
|
+
|
|
20
|
+
# Check each input
|
|
21
|
+
for idx, inp in enumerate(inputs):
|
|
22
|
+
analytical_grad = analytical_grads[idx]
|
|
23
|
+
|
|
24
|
+
# Compute numerical gradient for each element
|
|
25
|
+
for i in range(len(inp._data)):
|
|
26
|
+
original = inp._data[i]
|
|
27
|
+
|
|
28
|
+
# f(x + h)
|
|
29
|
+
inp._data[i] = original + eps
|
|
30
|
+
# Clear grads and recompute
|
|
31
|
+
for inp2 in inputs:
|
|
32
|
+
inp2.grad = None
|
|
33
|
+
inp2._grad_fn = None
|
|
34
|
+
out_plus = func(*inputs)
|
|
35
|
+
|
|
36
|
+
# f(x - h)
|
|
37
|
+
inp._data[i] = original - eps
|
|
38
|
+
for inp2 in inputs:
|
|
39
|
+
inp2.grad = None
|
|
40
|
+
inp2._grad_fn = None
|
|
41
|
+
out_minus = func(*inputs)
|
|
42
|
+
|
|
43
|
+
# Restore
|
|
44
|
+
inp._data[i] = original
|
|
45
|
+
|
|
46
|
+
# Numerical gradient
|
|
47
|
+
numerical = (out_plus._data[0] - out_minus._data[0]) / (2 * eps)
|
|
48
|
+
analytical = analytical_grad._data[i]
|
|
49
|
+
|
|
50
|
+
diff = abs(numerical - analytical)
|
|
51
|
+
if diff > tol:
|
|
52
|
+
raise AssertionError(
|
|
53
|
+
f"Gradient check failed for input {idx}, element {i}: "
|
|
54
|
+
f"numerical={numerical:.6f}, analytical={analytical:.6f}, "
|
|
55
|
+
f"diff={diff:.6f}"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
return True
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import array
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
import Metal
|
|
5
|
+
import MetalPerformanceShaders as MPS
|
|
6
|
+
|
|
7
|
+
_DEVICE = Metal.MTLCreateSystemDefaultDevice()
|
|
8
|
+
_QUEUE = _DEVICE.newCommandQueue() if _DEVICE is not None else None
|
|
9
|
+
|
|
10
|
+
MPS_AVAILABLE = _DEVICE is not None and _QUEUE is not None
|
|
11
|
+
if MPS_AVAILABLE:
|
|
12
|
+
_SHARED = Metal.MTLResourceStorageModeShared
|
|
13
|
+
except Exception:
|
|
14
|
+
_DEVICE = None
|
|
15
|
+
_QUEUE = None
|
|
16
|
+
MPS_AVAILABLE = False
|
|
17
|
+
|
|
18
|
+
_FLOAT_BYTES = 4
|
|
19
|
+
GPU_MIN_WORK = 200000
|
|
20
|
+
|
|
21
|
+
def _descriptor(rows, cols):
|
|
22
|
+
return MPS.MPSMatrixDescriptor.matrixDescriptorWithRows_columns_rowBytes_dataType_(
|
|
23
|
+
rows, cols, cols * _FLOAT_BYTES, MPS.MPSDataTypeFloat32
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
def mps_matmul(a_data, b_data, m, k, n):
|
|
27
|
+
buf_a = _DEVICE.newBufferWithBytes_length_options_(
|
|
28
|
+
a_data.tobytes(), m * k * _FLOAT_BYTES, _SHARED
|
|
29
|
+
)
|
|
30
|
+
buf_b = _DEVICE.newBufferWithBytes_length_options_(
|
|
31
|
+
b_data.tobytes(), k * n * _FLOAT_BYTES, _SHARED
|
|
32
|
+
)
|
|
33
|
+
buf_c = _DEVICE.newBufferWithLength_options_(m * n * _FLOAT_BYTES, _SHARED)
|
|
34
|
+
|
|
35
|
+
mat_a = MPS.MPSMatrix.alloc().initWithBuffer_descriptor_(buf_a, _descriptor(m, k))
|
|
36
|
+
mat_b = MPS.MPSMatrix.alloc().initWithBuffer_descriptor_(buf_b, _descriptor(k, n))
|
|
37
|
+
mat_c = MPS.MPSMatrix.alloc().initWithBuffer_descriptor_(buf_c, _descriptor(m, n))
|
|
38
|
+
|
|
39
|
+
kernel = MPS.MPSMatrixMultiplication.alloc().initWithDevice_transposeLeft_transposeRight_resultRows_resultColumns_interiorColumns_alpha_beta_(
|
|
40
|
+
_DEVICE, False, False, m, n, k, 1.0, 0.0
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
cmd = _QUEUE.commandBuffer()
|
|
44
|
+
kernel.encodeToCommandBuffer_leftMatrix_rightMatrix_resultMatrix_(
|
|
45
|
+
cmd, mat_a, mat_b, mat_c
|
|
46
|
+
)
|
|
47
|
+
cmd.commit()
|
|
48
|
+
cmd.waitUntilCompleted()
|
|
49
|
+
|
|
50
|
+
out = array.array('f')
|
|
51
|
+
out.frombytes(buf_c.contents().as_buffer(m * n * _FLOAT_BYTES))
|
|
52
|
+
return out
|