forge-dl 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
forge/__init__.py ADDED
@@ -0,0 +1,19 @@
1
+ """
2
+ Forge: a machine learning library and deep learning compiler written from
3
+ scratch in Python.
4
+
5
+ The core has no dependencies. Every operation, from the autograd engine to the
6
+ attention mechanism, is implemented from first principles.
7
+ """
8
+
9
+ from forge.tensor import Tensor
10
+ from forge.dtype import float32, float64, int32, int64
11
+ from forge.serialization import save_model, load_model
12
+
13
+ __version__ = "0.1.0"
14
+
15
+ __all__ = [
16
+ "Tensor",
17
+ "float32", "float64", "int32", "int64",
18
+ "save_model", "load_model",
19
+ ]
@@ -0,0 +1,20 @@
1
+ """
2
+ The autograd engine: reverse-mode automatic differentiation on a define-by-run
3
+ graph.
4
+
5
+ Every operation records itself as it executes, and backward() walks that record
6
+ in reverse, applying the chain rule at each node.
7
+ """
8
+
9
+ from forge.autograd.engine import Function
10
+ from forge.autograd.operations import (
11
+ Add, Mul, Sub, Pow, Neg, Sum, Mean, Matmul,
12
+ ReLU, Sigmoid, Tanh, Log, Clamp, Softmax, Exp,
13
+ )
14
+ from forge.autograd.grad_check import grad_check
15
+
16
+ __all__ = [
17
+ "Function", "grad_check",
18
+ "Add", "Mul", "Sub", "Pow", "Neg", "Sum", "Mean", "Matmul",
19
+ "ReLU", "Sigmoid", "Tanh", "Log", "Clamp", "Softmax", "Exp",
20
+ ]
@@ -0,0 +1,70 @@
1
+ import array
2
+ import ctypes
3
+
4
+ try:
5
+ _lib = ctypes.CDLL("/System/Library/Frameworks/Accelerate.framework/Accelerate")
6
+
7
+ cblas_sgemm = _lib.cblas_sgemm
8
+
9
+ float_pointer = ctypes.POINTER(ctypes.c_float)
10
+
11
+ cblas_sgemm.argtypes = [
12
+ ctypes.c_int, # storage order (row-major vs column-major)
13
+ ctypes.c_int, # whether to transpose the left matrix
14
+ ctypes.c_int, # whether to transpose the right matrix
15
+ ctypes.c_int, # number of rows in the result
16
+ ctypes.c_int, # number of columns in the result
17
+ ctypes.c_int, # the shared inner dimension being summed over
18
+ ctypes.c_float, # alpha: the product is scaled by this
19
+ float_pointer, # pointer to the left matrix's data
20
+ ctypes.c_int, # row stride of the left matrix
21
+ float_pointer, # pointer to the right matrix's data
22
+ ctypes.c_int, # row stride of the right matrix
23
+ ctypes.c_float, # beta: the existing result is scaled by this
24
+ float_pointer, # pointer to the output matrix's data
25
+ ctypes.c_int, # row stride of the output matrix
26
+ ]
27
+
28
+ cblas_sgemm.restype = None
29
+
30
+ ACCELERATE_AVAILABLE = True
31
+
32
+ # Only occurs on non-macOS systems
33
+ except Exception:
34
+ ACCELERATE_AVAILABLE = False
35
+
36
+ CBLAS_ROW_MAJOR = 101
37
+ CBLAS_NO_TRANSPOSE = 111
38
+
39
+ BYTES_PER_FLOAT = 4
40
+
41
+ def accelerate_matmul(left_data, right_data, left_rows, shared_dim, right_cols):
42
+ """
43
+ Multiplies two matrices on the CPU using Accelerate.
44
+
45
+ Returns a flat float32 array of the result.
46
+ """
47
+ result_data = array.array('f', bytes(BYTES_PER_FLOAT * left_rows * right_cols))
48
+
49
+ left_pointer = ctypes.cast(left_data.buffer_info()[0], float_pointer)
50
+ right_pointer = ctypes.cast(right_data.buffer_info()[0], float_pointer)
51
+ result_pointer = ctypes.cast(result_data.buffer_info()[0], float_pointer)
52
+
53
+ cblas_sgemm(
54
+ CBLAS_ROW_MAJOR, # data is laid out row by row
55
+ CBLAS_NO_TRANSPOSE, # do not transpose the left matrix
56
+ CBLAS_NO_TRANSPOSE, # do not transpose the right matrix
57
+ left_rows, # rows of the result
58
+ right_cols, # columns of the result
59
+ shared_dim, # the inner dimension summed over
60
+ 1.0, # alpha: scale the product by 1.0
61
+ left_pointer, # the left matrix
62
+ shared_dim, # left's row stride
63
+ right_pointer, # the right matrix
64
+ right_cols, # right's row stride
65
+ 0.0, # beta: ignore the zeroed existing result
66
+ result_pointer, # where to write the answer
67
+ right_cols, # result's row stride
68
+ )
69
+
70
+ return result_data
@@ -0,0 +1,14 @@
1
+ class Function:
2
+ """Base class for all differentiable operations"""
3
+ def __init__(self):
4
+ self.inputs = []
5
+ self.saved_tensors = []
6
+
7
+ def save_for_backward(self, *tensors):
8
+ self.saved_tensors = list(tensors)
9
+
10
+ def forward(self, *args):
11
+ raise NotImplementedError
12
+
13
+ def backward(self, grad_output):
14
+ raise NotImplementedError
@@ -0,0 +1,120 @@
1
+ import array as _array
2
+ import math
3
+ from forge.autograd.engine import Function
4
+
5
+
6
+ class FusedLinearReLU(Function):
7
+ """Fused matmul + bias add + ReLU in a single pass"""
8
+
9
+ def forward(self, x, weight_t, bias):
10
+ self.inputs = [x, weight_t, bias]
11
+ self.save_for_backward(x, weight_t, bias)
12
+ from forge.tensor import Tensor
13
+
14
+ m = x.shape[0]
15
+ n = x.shape[1]
16
+ p = weight_t.shape[1]
17
+
18
+ new_data = _array.array(x.dtype.typecode, [])
19
+
20
+ for i in range(m):
21
+ for j in range(p):
22
+ # Matmul
23
+ total = 0.0
24
+ for k in range(n):
25
+ total += x._data[i * n + k] * weight_t._data[k * p + j]
26
+
27
+ # Add bias
28
+ total += bias._data[j]
29
+
30
+ # ReLU
31
+ if total < 0:
32
+ total = 0.0
33
+
34
+ new_data.append(total)
35
+
36
+ result = Tensor.__new__(Tensor)
37
+ result._data = new_data
38
+ result.shape = (m, p)
39
+ result.dtype = x.dtype
40
+ result.requires_grad = False
41
+ result.grad = None
42
+ result._grad_fn = None
43
+
44
+ # Save output for backward (need to know where ReLU killed values)
45
+ self._output = result
46
+ return result
47
+
48
+ def backward(self, grad_output):
49
+ x, weight_t, bias = self.saved_tensors
50
+ output = self._output
51
+ from forge.tensor import Tensor, _broadcast_shape, _broadcast_data
52
+ from forge.autograd.operations import _unbroadcast
53
+
54
+ m = x.shape[0]
55
+ n = x.shape[1]
56
+ p = weight_t.shape[1]
57
+
58
+ # Apply ReLU mask to grad_output
59
+ relu_grad_data = _array.array(
60
+ grad_output.dtype.typecode,
61
+ [g if o > 0 else 0.0 for g, o in zip(grad_output._data, output._data)]
62
+ )
63
+ relu_grad = Tensor.__new__(Tensor)
64
+ relu_grad._data = relu_grad_data
65
+ relu_grad.shape = grad_output.shape
66
+ relu_grad.dtype = grad_output.dtype
67
+ relu_grad.requires_grad = False
68
+ relu_grad.grad = None
69
+ relu_grad._grad_fn = None
70
+
71
+ # Grad for x: relu_grad @ weight_t.T
72
+ # weight_t is (n, p), weight_t.T is (p, n)
73
+ grad_x_data = _array.array(x.dtype.typecode, [0.0] * (m * n))
74
+ for i in range(m):
75
+ for j in range(n):
76
+ total = 0.0
77
+ for k in range(p):
78
+ total += relu_grad._data[i * p + k] * weight_t._data[j * p + k]
79
+ grad_x_data[i * n + j] = total
80
+
81
+ grad_x = Tensor.__new__(Tensor)
82
+ grad_x._data = grad_x_data
83
+ grad_x.shape = x.shape
84
+ grad_x.dtype = x.dtype
85
+ grad_x.requires_grad = False
86
+ grad_x.grad = None
87
+ grad_x._grad_fn = None
88
+
89
+ # Grad for weight_t: x.T @ relu_grad
90
+ grad_wt_data = _array.array(x.dtype.typecode, [0.0] * (n * p))
91
+ for i in range(n):
92
+ for j in range(p):
93
+ total = 0.0
94
+ for k in range(m):
95
+ total += x._data[k * n + i] * relu_grad._data[k * p + j]
96
+ grad_wt_data[i * p + j] = total
97
+
98
+ grad_wt = Tensor.__new__(Tensor)
99
+ grad_wt._data = grad_wt_data
100
+ grad_wt.shape = weight_t.shape
101
+ grad_wt.dtype = weight_t.dtype
102
+ grad_wt.requires_grad = False
103
+ grad_wt.grad = None
104
+ grad_wt._grad_fn = None
105
+
106
+ # Grad for bias: sum relu_grad along rows
107
+ grad_bias_data = _array.array(bias.dtype.typecode, [0.0] * p)
108
+ for i in range(m):
109
+ for j in range(p):
110
+ grad_bias_data[j] += relu_grad._data[i * p + j]
111
+
112
+ grad_bias = Tensor.__new__(Tensor)
113
+ grad_bias._data = grad_bias_data
114
+ grad_bias.shape = bias.shape
115
+ grad_bias.dtype = bias.dtype
116
+ grad_bias.requires_grad = False
117
+ grad_bias.grad = None
118
+ grad_bias._grad_fn = None
119
+
120
+ return grad_x, grad_wt, grad_bias
@@ -0,0 +1,40 @@
1
+ from forge.nn.layers import Linear, ReLULayer, FusedLinearReLULayer
2
+
3
+
4
+ def optimize_model(layers):
5
+ """
6
+ Takes a list of layers and returns an optimized list
7
+ with fused operations where possible.
8
+ """
9
+ optimized = []
10
+ i = 0
11
+
12
+ while i < len(layers):
13
+ # Pattern: Linear followed by ReLU
14
+ if (i + 1 < len(layers) and isinstance(layers[i], Linear) and isinstance(layers[i + 1], ReLULayer)):
15
+ linear = layers[i]
16
+ in_f = linear.weight.shape[1]
17
+ out_f = linear.weight.shape[0]
18
+
19
+ fused = FusedLinearReLULayer(in_f, out_f)
20
+
21
+ # Copy weights from the original linear layer
22
+ fused.weight._data = linear.weight._data.__class__(
23
+ linear.weight.dtype.typecode, linear.weight._data
24
+ )
25
+
26
+ # Copy bias — need to flatten from (1, out_f) to (out_f,)
27
+ if linear.bias is not None:
28
+ import array as _array
29
+ fused.bias._data = _array.array(
30
+ linear.bias.dtype.typecode, list(linear.bias._data)
31
+ )
32
+
33
+ optimized.append(fused)
34
+ i += 2 # Skip both the Linear and ReLU
35
+
36
+ else:
37
+ optimized.append(layers[i])
38
+ i += 1
39
+
40
+ return optimized
@@ -0,0 +1,58 @@
1
+ """
2
+ It is important to draw wisdom from many different places.
3
+ If you take it from only one place, it becomes rigid and stale.
4
+ - Uncle Iroh
5
+ """
6
+
7
+ def grad_check(func, inputs, eps=1e-3, tol=1e-2):
8
+ """
9
+ Using first principles for differentiation: df/dx = lim(h->0) [f(x + h) - f(x - h)] / 2h
10
+ We use a small value of h to approximate the limit.
11
+ """
12
+ for inp in inputs:
13
+ inp.grad = None
14
+
15
+ output = func(*inputs)
16
+ output.backward()
17
+
18
+ analytical_grads = [inp.grad for inp in inputs]
19
+
20
+ # Check each input
21
+ for idx, inp in enumerate(inputs):
22
+ analytical_grad = analytical_grads[idx]
23
+
24
+ # Compute numerical gradient for each element
25
+ for i in range(len(inp._data)):
26
+ original = inp._data[i]
27
+
28
+ # f(x + h)
29
+ inp._data[i] = original + eps
30
+ # Clear grads and recompute
31
+ for inp2 in inputs:
32
+ inp2.grad = None
33
+ inp2._grad_fn = None
34
+ out_plus = func(*inputs)
35
+
36
+ # f(x - h)
37
+ inp._data[i] = original - eps
38
+ for inp2 in inputs:
39
+ inp2.grad = None
40
+ inp2._grad_fn = None
41
+ out_minus = func(*inputs)
42
+
43
+ # Restore
44
+ inp._data[i] = original
45
+
46
+ # Numerical gradient
47
+ numerical = (out_plus._data[0] - out_minus._data[0]) / (2 * eps)
48
+ analytical = analytical_grad._data[i]
49
+
50
+ diff = abs(numerical - analytical)
51
+ if diff > tol:
52
+ raise AssertionError(
53
+ f"Gradient check failed for input {idx}, element {i}: "
54
+ f"numerical={numerical:.6f}, analytical={analytical:.6f}, "
55
+ f"diff={diff:.6f}"
56
+ )
57
+
58
+ return True
@@ -0,0 +1,52 @@
1
+ import array
2
+
3
+ try:
4
+ import Metal
5
+ import MetalPerformanceShaders as MPS
6
+
7
+ _DEVICE = Metal.MTLCreateSystemDefaultDevice()
8
+ _QUEUE = _DEVICE.newCommandQueue() if _DEVICE is not None else None
9
+
10
+ MPS_AVAILABLE = _DEVICE is not None and _QUEUE is not None
11
+ if MPS_AVAILABLE:
12
+ _SHARED = Metal.MTLResourceStorageModeShared
13
+ except Exception:
14
+ _DEVICE = None
15
+ _QUEUE = None
16
+ MPS_AVAILABLE = False
17
+
18
+ _FLOAT_BYTES = 4
19
+ GPU_MIN_WORK = 200000
20
+
21
+ def _descriptor(rows, cols):
22
+ return MPS.MPSMatrixDescriptor.matrixDescriptorWithRows_columns_rowBytes_dataType_(
23
+ rows, cols, cols * _FLOAT_BYTES, MPS.MPSDataTypeFloat32
24
+ )
25
+
26
+ def mps_matmul(a_data, b_data, m, k, n):
27
+ buf_a = _DEVICE.newBufferWithBytes_length_options_(
28
+ a_data.tobytes(), m * k * _FLOAT_BYTES, _SHARED
29
+ )
30
+ buf_b = _DEVICE.newBufferWithBytes_length_options_(
31
+ b_data.tobytes(), k * n * _FLOAT_BYTES, _SHARED
32
+ )
33
+ buf_c = _DEVICE.newBufferWithLength_options_(m * n * _FLOAT_BYTES, _SHARED)
34
+
35
+ mat_a = MPS.MPSMatrix.alloc().initWithBuffer_descriptor_(buf_a, _descriptor(m, k))
36
+ mat_b = MPS.MPSMatrix.alloc().initWithBuffer_descriptor_(buf_b, _descriptor(k, n))
37
+ mat_c = MPS.MPSMatrix.alloc().initWithBuffer_descriptor_(buf_c, _descriptor(m, n))
38
+
39
+ kernel = MPS.MPSMatrixMultiplication.alloc().initWithDevice_transposeLeft_transposeRight_resultRows_resultColumns_interiorColumns_alpha_beta_(
40
+ _DEVICE, False, False, m, n, k, 1.0, 0.0
41
+ )
42
+
43
+ cmd = _QUEUE.commandBuffer()
44
+ kernel.encodeToCommandBuffer_leftMatrix_rightMatrix_resultMatrix_(
45
+ cmd, mat_a, mat_b, mat_c
46
+ )
47
+ cmd.commit()
48
+ cmd.waitUntilCompleted()
49
+
50
+ out = array.array('f')
51
+ out.frombytes(buf_c.contents().as_buffer(m * n * _FLOAT_BYTES))
52
+ return out