pytensorforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli.py +604 -0
- pytensorforge-0.1.0.dist-info/METADATA +103 -0
- pytensorforge-0.1.0.dist-info/RECORD +146 -0
- pytensorforge-0.1.0.dist-info/WHEEL +5 -0
- pytensorforge-0.1.0.dist-info/entry_points.txt +2 -0
- pytensorforge-0.1.0.dist-info/top_level.txt +2 -0
- src/__init__.py +0 -0
- src/activations/Activation.py +4 -0
- src/activations/ELU.py +11 -0
- src/activations/GELU.py +6 -0
- src/activations/ReLU.py +27 -0
- src/activations/SELU.py +14 -0
- src/activations/Sigmoid.py +27 -0
- src/activations/Softmax.py +84 -0
- src/activations/Tanh.py +29 -0
- src/activations/__init__.py +17 -0
- src/config.py +120 -0
- src/core/Matrix.py +3 -0
- src/core/Scalar.py +18 -0
- src/core/Tensor.py +866 -0
- src/core/Vector.py +31 -0
- src/core/__init__.py +0 -0
- src/data/__init__.py +0 -0
- src/data/chat_dataset.py +188 -0
- src/data/corpus.py +104 -0
- src/data/document_stream.py +178 -0
- src/data/parallel_encode.py +86 -0
- src/data/prefetch.py +62 -0
- src/data/shard_builder.py +119 -0
- src/data/shard_writer.py +81 -0
- src/data/sharded_dataset.py +112 -0
- src/data/streaming_dataset.py +132 -0
- src/data/validation.py +212 -0
- src/inference/__init__.py +0 -0
- src/inference/chat_template.py +384 -0
- src/inference/config.py +48 -0
- src/inference/engine.py +241 -0
- src/inference/export.py +133 -0
- src/inference/kv_cache.py +65 -0
- src/inference/runtime.py +161 -0
- src/inference/sampling.py +42 -0
- src/inference/scheduler.py +473 -0
- src/inference/text.py +67 -0
- src/initializers/Constant.py +9 -0
- src/initializers/GlorotNormal.py +15 -0
- src/initializers/GlorotUniform.py +26 -0
- src/initializers/HeNormal.py +15 -0
- src/initializers/HeUniform.py +14 -0
- src/initializers/Initializer.py +4 -0
- src/initializers/LecunNormal.py +16 -0
- src/initializers/LecunUniform.py +14 -0
- src/initializers/Ones.py +6 -0
- src/initializers/Orthogonal.py +14 -0
- src/initializers/RandomNormal.py +14 -0
- src/initializers/RandomUniform.py +14 -0
- src/initializers/Zeros.py +8 -0
- src/initializers/__init__.py +17 -0
- src/loss/CategoricalCrossEntropy.py +9 -0
- src/loss/CrossEntropyLoss.py +34 -0
- src/loss/CrossEntropyWithLogitsLoss.py +59 -0
- src/loss/Hinge.py +5 -0
- src/loss/Huber.py +22 -0
- src/loss/Loss.py +6 -0
- src/loss/MSE.py +7 -0
- src/loss/MSELoss.py +10 -0
- src/loss/SparseCategoricalCrossEntropy.py +15 -0
- src/loss/__init__.py +18 -0
- src/loss/bce.py +34 -0
- src/loss/mae.py +16 -0
- src/math/__init__.py +0 -0
- src/math/clip.py +37 -0
- src/math/exp.py +27 -0
- src/math/log.py +25 -0
- src/math/sigmoid.py +5 -0
- src/models/__init__.py +0 -0
- src/models/embedding/Embedding.py +65 -0
- src/models/embedding/__init__.py +0 -0
- src/models/gpt/__init__.py +0 -0
- src/models/gpt/attention.py +158 -0
- src/models/gpt/block.py +74 -0
- src/models/gpt/config.py +103 -0
- src/models/gpt/context.py +44 -0
- src/models/gpt/model.py +165 -0
- src/models/gpt/recompute.py +35 -0
- src/models/gpt/rope.py +84 -0
- src/models/regression/Linear.py +51 -0
- src/models/regression/Logistic.py +36 -0
- src/models/regression/__init__.py +0 -0
- src/models/seq/Sequential.py +297 -0
- src/models/seq/__init__.py +0 -0
- src/models/svm/__init__.py +0 -0
- src/models/tokenizer/BPETokenizer.py +228 -0
- src/models/tokenizer/__init__.py +0 -0
- src/models/transformers/Dropout.py +35 -0
- src/models/transformers/LastToken.py +10 -0
- src/models/transformers/LayerNorm.py +54 -0
- src/models/transformers/Linear.py +18 -0
- src/models/transformers/MultiHeadAttention.py +130 -0
- src/models/transformers/TransformerBlock.py +79 -0
- src/models/transformers/__init__.py +0 -0
- src/neural/Dense.py +58 -0
- src/neural/LSTM.py +167 -0
- src/neural/Layer.py +72 -0
- src/neural/Parameter.py +30 -0
- src/neural/RNN.py +83 -0
- src/neural/__init__.py +0 -0
- src/ops/__init__.py +0 -0
- src/ops/stack.py +40 -0
- src/optimizers/Adagrad.py +31 -0
- src/optimizers/Adam.py +98 -0
- src/optimizers/AdamW.py +84 -0
- src/optimizers/Batch.py +11 -0
- src/optimizers/Nesterov.py +35 -0
- src/optimizers/Optimizer.py +18 -0
- src/optimizers/RMSProp.py +35 -0
- src/optimizers/SGD.py +30 -0
- src/optimizers/SGDMomentum.py +28 -0
- src/optimizers/__init__.py +9 -0
- src/scaling/StandardScaler.py +15 -0
- src/scaling/__init__.py +0 -0
- src/serialization/__init__.py +0 -0
- src/serialization/checkpoint.py +58 -0
- src/serialization/modelio.py +132 -0
- src/serving/__init__.py +0 -0
- src/serving/app.py +792 -0
- src/serving/config.py +216 -0
- src/serving/errors.py +51 -0
- src/serving/http.py +599 -0
- src/serving/metrics.py +293 -0
- src/serving/model_server.py +287 -0
- src/serving/protocol.py +377 -0
- src/serving/security.py +200 -0
- src/serving/server.py +121 -0
- src/tokenization/__init__.py +0 -0
- src/tokenization/base.py +75 -0
- src/tokenization/bpe.py +190 -0
- src/tokenization/bytebpe.py +476 -0
- src/tokenization/registry.py +28 -0
- src/training/__init__.py +0 -0
- src/training/checkpoint_manager.py +101 -0
- src/training/experiment.py +71 -0
- src/training/losses.py +42 -0
- src/training/precision.py +141 -0
- src/training/profiler.py +38 -0
- src/training/scheduler.py +50 -0
- src/training/trainer.py +594 -0
src/core/Tensor.py
ADDED
|
@@ -0,0 +1,866 @@
|
|
|
1
|
+
import threading
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
_state = threading.local()
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def _noop():
|
|
9
|
+
return None
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def grad_enabled():
|
|
13
|
+
return getattr(_state, "grad_enabled", True)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def matmul_policy():
|
|
17
|
+
return getattr(_state, "matmul_policy", None)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def set_matmul_policy(fn):
|
|
21
|
+
previous = matmul_policy()
|
|
22
|
+
_state.matmul_policy = fn
|
|
23
|
+
return previous
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class no_grad:
|
|
27
|
+
|
|
28
|
+
def __enter__(self):
|
|
29
|
+
self._previous = grad_enabled()
|
|
30
|
+
_state.grad_enabled = False
|
|
31
|
+
return self
|
|
32
|
+
|
|
33
|
+
def __exit__(self, *exc):
|
|
34
|
+
_state.grad_enabled = self._previous
|
|
35
|
+
return False
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _round_operand(x):
|
|
39
|
+
policy = matmul_policy()
|
|
40
|
+
return x if policy is None else policy(x)
|
|
41
|
+
|
|
42
|
+
# x = Tensor([[1, 2],
|
|
43
|
+
# [3, 4]], requires_grad=True)
|
|
44
|
+
#
|
|
45
|
+
# print(x.shape)
|
|
46
|
+
|
|
47
|
+
def unbroadcast(grad, shape):
|
|
48
|
+
while grad.ndim > len(shape):
|
|
49
|
+
grad = grad.sum(axis=0)
|
|
50
|
+
|
|
51
|
+
for axis, size in enumerate(shape):
|
|
52
|
+
if size == 1:
|
|
53
|
+
grad = grad.sum(axis=axis, keepdims=True)
|
|
54
|
+
|
|
55
|
+
return grad
|
|
56
|
+
|
|
57
|
+
def _topological_order(root):
|
|
58
|
+
order = []
|
|
59
|
+
visited = {id(root)}
|
|
60
|
+
stack = [(root, iter(root.parents))]
|
|
61
|
+
|
|
62
|
+
while stack:
|
|
63
|
+
node, children = stack[-1]
|
|
64
|
+
advanced = False
|
|
65
|
+
|
|
66
|
+
for child in children:
|
|
67
|
+
if id(child) not in visited:
|
|
68
|
+
visited.add(id(child))
|
|
69
|
+
stack.append((child, iter(child.parents)))
|
|
70
|
+
advanced = True
|
|
71
|
+
break
|
|
72
|
+
|
|
73
|
+
if not advanced:
|
|
74
|
+
stack.pop()
|
|
75
|
+
order.append(node)
|
|
76
|
+
|
|
77
|
+
return order
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class Tensor:
|
|
81
|
+
|
|
82
|
+
def __init__(
|
|
83
|
+
self,
|
|
84
|
+
data,
|
|
85
|
+
requires_grad=True,
|
|
86
|
+
parents=(),
|
|
87
|
+
op=None,
|
|
88
|
+
):
|
|
89
|
+
self.data = np.asarray(data, dtype=np.float32)
|
|
90
|
+
|
|
91
|
+
self._grad = None
|
|
92
|
+
|
|
93
|
+
if not grad_enabled():
|
|
94
|
+
requires_grad = False
|
|
95
|
+
parents = ()
|
|
96
|
+
|
|
97
|
+
self.requires_grad = requires_grad
|
|
98
|
+
|
|
99
|
+
self.parents = parents
|
|
100
|
+
|
|
101
|
+
self.op = op
|
|
102
|
+
|
|
103
|
+
self._bw = _noop
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def grad(self):
|
|
107
|
+
if self._grad is None:
|
|
108
|
+
self._grad = np.zeros_like(self.data)
|
|
109
|
+
return self._grad
|
|
110
|
+
|
|
111
|
+
@grad.setter
|
|
112
|
+
def grad(self, value):
|
|
113
|
+
self._grad = value
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def _backward(self):
|
|
117
|
+
return self._bw
|
|
118
|
+
|
|
119
|
+
@_backward.setter
|
|
120
|
+
def _backward(self, fn):
|
|
121
|
+
self._bw = fn if self.requires_grad else _noop
|
|
122
|
+
|
|
123
|
+
@property
|
|
124
|
+
def shape(self):
|
|
125
|
+
return self.data.shape
|
|
126
|
+
|
|
127
|
+
def zero_grad(self):
|
|
128
|
+
if self._grad is not None:
|
|
129
|
+
self._grad.fill(0)
|
|
130
|
+
|
|
131
|
+
def __len__(self):
|
|
132
|
+
return self.data.shape[0]
|
|
133
|
+
|
|
134
|
+
# def __getitem__(self, idx):
|
|
135
|
+
# return Tensor(
|
|
136
|
+
# self.data[idx],
|
|
137
|
+
# requires_grad=self.requires_grad,
|
|
138
|
+
# )
|
|
139
|
+
|
|
140
|
+
def __getitem__(self, idx):
|
|
141
|
+
|
|
142
|
+
if isinstance(idx, Tensor):
|
|
143
|
+
idx = idx.data.astype(np.int64)
|
|
144
|
+
|
|
145
|
+
elif isinstance(idx, tuple):
|
|
146
|
+
idx = tuple(
|
|
147
|
+
i.data.astype(np.int64) if isinstance(i, Tensor) else i
|
|
148
|
+
for i in idx
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
out = Tensor(
|
|
152
|
+
self.data[idx],
|
|
153
|
+
requires_grad=self.requires_grad,
|
|
154
|
+
parents=(self,),
|
|
155
|
+
op="Slice",
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
def _backward():
|
|
159
|
+
if not self.requires_grad:
|
|
160
|
+
return
|
|
161
|
+
|
|
162
|
+
np.add.at(
|
|
163
|
+
self.grad,
|
|
164
|
+
idx,
|
|
165
|
+
out.grad,
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
out._backward = _backward
|
|
169
|
+
|
|
170
|
+
return out
|
|
171
|
+
|
|
172
|
+
@staticmethod
|
|
173
|
+
def zeros(shape, requires_grad=False):
|
|
174
|
+
return Tensor(
|
|
175
|
+
np.zeros(shape, dtype=np.float32),
|
|
176
|
+
requires_grad=requires_grad,
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
@staticmethod
|
|
180
|
+
def stack(tensors, axis=0):
|
|
181
|
+
from src.ops.stack import Stack
|
|
182
|
+
return Stack.forward(tensors, axis)
|
|
183
|
+
|
|
184
|
+
def __repr__(self):
|
|
185
|
+
return f"Tensor(data={self.data}, requires_grad={self.requires_grad})"
|
|
186
|
+
|
|
187
|
+
def __add__(self, other):
|
|
188
|
+
if not isinstance(other, Tensor):
|
|
189
|
+
other = Tensor(other)
|
|
190
|
+
|
|
191
|
+
out = Tensor(
|
|
192
|
+
self.data + other.data,
|
|
193
|
+
requires_grad=self.requires_grad or other.requires_grad,
|
|
194
|
+
parents=(self, other),
|
|
195
|
+
op="Add",
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
def backward():
|
|
199
|
+
if self.requires_grad:
|
|
200
|
+
self.grad += unbroadcast(out.grad, self.shape)
|
|
201
|
+
|
|
202
|
+
if other.requires_grad:
|
|
203
|
+
other.grad += unbroadcast(out.grad, other.shape)
|
|
204
|
+
|
|
205
|
+
out._backward = backward
|
|
206
|
+
|
|
207
|
+
return out
|
|
208
|
+
|
|
209
|
+
def __sub__(self, other):
|
|
210
|
+
|
|
211
|
+
if not isinstance(other, Tensor):
|
|
212
|
+
other = Tensor(other)
|
|
213
|
+
|
|
214
|
+
out = Tensor(
|
|
215
|
+
self.data - other.data,
|
|
216
|
+
requires_grad=self.requires_grad or other.requires_grad,
|
|
217
|
+
parents=(self, other),
|
|
218
|
+
op="Sub",
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
def _backward_():
|
|
222
|
+
# out = self - other
|
|
223
|
+
# 𝜹out/𝜹self = 1 - 0 = 1
|
|
224
|
+
self.grad += out.grad * 1
|
|
225
|
+
# 𝜹out/𝜹other = 0 - 1 = -1
|
|
226
|
+
other.grad += out.grad * (-1)
|
|
227
|
+
|
|
228
|
+
def _backward():
|
|
229
|
+
if self.requires_grad:
|
|
230
|
+
self.grad += unbroadcast(out.grad, self.shape)
|
|
231
|
+
|
|
232
|
+
if other.requires_grad:
|
|
233
|
+
other.grad += unbroadcast(-out.grad, other.shape)
|
|
234
|
+
|
|
235
|
+
out._backward = _backward
|
|
236
|
+
return out
|
|
237
|
+
|
|
238
|
+
# def __neg__(self):
|
|
239
|
+
# return self * -1
|
|
240
|
+
|
|
241
|
+
def __neg__(self):
|
|
242
|
+
out = Tensor(
|
|
243
|
+
-self.data,
|
|
244
|
+
requires_grad=self.requires_grad,
|
|
245
|
+
parents=(self,),
|
|
246
|
+
op="Neg",
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
def _backward():
|
|
250
|
+
if self.requires_grad:
|
|
251
|
+
self.grad -= out.grad
|
|
252
|
+
|
|
253
|
+
out._backward = _backward
|
|
254
|
+
|
|
255
|
+
return out
|
|
256
|
+
|
|
257
|
+
def __mul__(self, other):
|
|
258
|
+
if not isinstance(other, Tensor):
|
|
259
|
+
other = Tensor(other)
|
|
260
|
+
|
|
261
|
+
out = Tensor(
|
|
262
|
+
self.data * other.data,
|
|
263
|
+
requires_grad=self.requires_grad or other.requires_grad,
|
|
264
|
+
parents=(self, other),
|
|
265
|
+
op="Mul",
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
def _backward_():
|
|
269
|
+
# out = self * other
|
|
270
|
+
# 𝜹out/𝜹self = other
|
|
271
|
+
self.grad += out.grad * other.data
|
|
272
|
+
# out = self * other
|
|
273
|
+
# 𝜹out/𝜹other = self
|
|
274
|
+
other.grad += out.grad * self.data
|
|
275
|
+
|
|
276
|
+
def _backward():
|
|
277
|
+
if self.requires_grad:
|
|
278
|
+
self.grad += unbroadcast(
|
|
279
|
+
out.grad * other.data,
|
|
280
|
+
self.shape,
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
if other.requires_grad:
|
|
284
|
+
other.grad += unbroadcast(
|
|
285
|
+
out.grad * self.data,
|
|
286
|
+
other.shape,
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
out._backward = _backward
|
|
290
|
+
return out
|
|
291
|
+
|
|
292
|
+
# def __matmul__(self, other):
|
|
293
|
+
# if not isinstance(other, Tensor):
|
|
294
|
+
# other = Tensor(other)
|
|
295
|
+
#
|
|
296
|
+
# out = Tensor(self.data @ other.data, requires_grad=self.requires_grad or other.requires_grad, parents=(self, other), op="MatMul")
|
|
297
|
+
# def _backward():
|
|
298
|
+
# self.grad += out.grad @ other.data.T
|
|
299
|
+
# other.grad += self.data.T @ out.grad
|
|
300
|
+
# out._backward = _backward
|
|
301
|
+
# return out
|
|
302
|
+
|
|
303
|
+
def __matmul__(self, other):
|
|
304
|
+
|
|
305
|
+
if not isinstance(other, Tensor):
|
|
306
|
+
other = Tensor(other)
|
|
307
|
+
|
|
308
|
+
policy = matmul_policy()
|
|
309
|
+
a = self.data if policy is None else policy(self.data)
|
|
310
|
+
b = other.data if policy is None else policy(other.data)
|
|
311
|
+
result = a @ b
|
|
312
|
+
|
|
313
|
+
out = Tensor(
|
|
314
|
+
result if policy is None else policy(result),
|
|
315
|
+
requires_grad=self.requires_grad or other.requires_grad,
|
|
316
|
+
parents=(self, other),
|
|
317
|
+
op="MatMul",
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
def _backward():
|
|
321
|
+
g = out.grad if policy is None else policy(out.grad)
|
|
322
|
+
|
|
323
|
+
if self.requires_grad:
|
|
324
|
+
grad = np.matmul(
|
|
325
|
+
g,
|
|
326
|
+
np.swapaxes(b, -1, -2),
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
if policy is not None:
|
|
330
|
+
grad = policy(grad)
|
|
331
|
+
|
|
332
|
+
self.grad += Tensor.unbroadcast(
|
|
333
|
+
grad,
|
|
334
|
+
self.shape,
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
if other.requires_grad:
|
|
338
|
+
grad = np.matmul(
|
|
339
|
+
np.swapaxes(a, -1, -2),
|
|
340
|
+
g,
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
if policy is not None:
|
|
344
|
+
grad = policy(grad)
|
|
345
|
+
|
|
346
|
+
other.grad += Tensor.unbroadcast(
|
|
347
|
+
grad,
|
|
348
|
+
other.shape,
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
out._backward = _backward
|
|
352
|
+
|
|
353
|
+
return out
|
|
354
|
+
|
|
355
|
+
@staticmethod
|
|
356
|
+
def unbroadcast(grad, shape):
|
|
357
|
+
while grad.ndim > len(shape):
|
|
358
|
+
grad = grad.sum(axis=0)
|
|
359
|
+
|
|
360
|
+
for axis, size in enumerate(shape):
|
|
361
|
+
if size == 1:
|
|
362
|
+
grad = grad.sum(axis=axis, keepdims=True)
|
|
363
|
+
|
|
364
|
+
return grad
|
|
365
|
+
|
|
366
|
+
def __div__(self, other):
|
|
367
|
+
if not isinstance(other, Tensor):
|
|
368
|
+
other = Tensor(other)
|
|
369
|
+
|
|
370
|
+
out = Tensor(
|
|
371
|
+
self.data / other.data,
|
|
372
|
+
requires_grad=self.requires_grad or other.requires_grad,
|
|
373
|
+
parents=(self, other),
|
|
374
|
+
op="Div",
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
def _backward():
|
|
378
|
+
# out = self / other
|
|
379
|
+
# 𝜹out/𝜹self = 1/other
|
|
380
|
+
# 𝜹out/𝜹other = self * (-1) * other ** (-2)
|
|
381
|
+
self.grad += out.grad * ( 1 / other.data)
|
|
382
|
+
other.grad += out.grad * (-self.data / (other.data ** 2))
|
|
383
|
+
out._backward = _backward
|
|
384
|
+
return out
|
|
385
|
+
|
|
386
|
+
# def __pow__(self, other):
|
|
387
|
+
# if not isinstance(other, Tensor):
|
|
388
|
+
# other = Tensor(other)
|
|
389
|
+
#
|
|
390
|
+
# out = Tensor(
|
|
391
|
+
# self.data ** other,
|
|
392
|
+
# requires_grad=self.requires_grad,
|
|
393
|
+
# parents=(self, ),
|
|
394
|
+
# op="Pow",
|
|
395
|
+
# )
|
|
396
|
+
#
|
|
397
|
+
# def _backward():
|
|
398
|
+
# # out = self ** other
|
|
399
|
+
# # 𝜹out/𝜹self = other * self ** (other - 1)
|
|
400
|
+
# self.grad += out.grad * (other * self.data ** (other - 1))
|
|
401
|
+
# out._backward = _backward
|
|
402
|
+
# return out
|
|
403
|
+
|
|
404
|
+
def __pow__(self, other):
|
|
405
|
+
|
|
406
|
+
if isinstance(other, Tensor):
|
|
407
|
+
other = other.data
|
|
408
|
+
|
|
409
|
+
out = Tensor(
|
|
410
|
+
self.data ** other,
|
|
411
|
+
requires_grad=self.requires_grad,
|
|
412
|
+
parents=(self,),
|
|
413
|
+
op="Pow",
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
def _backward():
|
|
417
|
+
if self.requires_grad:
|
|
418
|
+
self.grad += out.grad * other * (self.data ** (other - 1))
|
|
419
|
+
|
|
420
|
+
out._backward = _backward
|
|
421
|
+
|
|
422
|
+
return out
|
|
423
|
+
|
|
424
|
+
def __rpow__(self, other):
|
|
425
|
+
|
|
426
|
+
out = Tensor(
|
|
427
|
+
other ** self.data,
|
|
428
|
+
requires_grad=self.requires_grad,
|
|
429
|
+
parents=(self,),
|
|
430
|
+
op="RPow",
|
|
431
|
+
)
|
|
432
|
+
|
|
433
|
+
def _backward():
|
|
434
|
+
if self.requires_grad:
|
|
435
|
+
self.grad += (
|
|
436
|
+
out.grad
|
|
437
|
+
* np.log(other)
|
|
438
|
+
* (other ** self.data)
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
out._backward = _backward
|
|
442
|
+
|
|
443
|
+
return out
|
|
444
|
+
|
|
445
|
+
def sqrt(self):
|
|
446
|
+
|
|
447
|
+
out = Tensor(
|
|
448
|
+
np.sqrt(self.data),
|
|
449
|
+
requires_grad=self.requires_grad,
|
|
450
|
+
parents=(self,),
|
|
451
|
+
op="Sqrt",
|
|
452
|
+
)
|
|
453
|
+
|
|
454
|
+
def _backward():
|
|
455
|
+
if not self.requires_grad:
|
|
456
|
+
return
|
|
457
|
+
|
|
458
|
+
self.grad += out.grad * (0.5 / np.sqrt(self.data))
|
|
459
|
+
|
|
460
|
+
out._backward = _backward
|
|
461
|
+
|
|
462
|
+
return out
|
|
463
|
+
|
|
464
|
+
def masked_fill(self, mask, value):
|
|
465
|
+
|
|
466
|
+
if isinstance(mask, Tensor):
|
|
467
|
+
mask = mask.data
|
|
468
|
+
|
|
469
|
+
mask = np.broadcast_to(mask, self.data.shape)
|
|
470
|
+
|
|
471
|
+
out_data = self.data.copy()
|
|
472
|
+
|
|
473
|
+
out_data[mask] = value
|
|
474
|
+
|
|
475
|
+
out = Tensor(
|
|
476
|
+
out_data,
|
|
477
|
+
requires_grad=self.requires_grad,
|
|
478
|
+
parents=(self,),
|
|
479
|
+
op="MaskedFill",
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
def _backward():
|
|
483
|
+
|
|
484
|
+
if not self.requires_grad:
|
|
485
|
+
return
|
|
486
|
+
|
|
487
|
+
grad = out.grad.copy()
|
|
488
|
+
|
|
489
|
+
grad[mask] = 0
|
|
490
|
+
|
|
491
|
+
self.grad += grad
|
|
492
|
+
|
|
493
|
+
out._backward = _backward
|
|
494
|
+
|
|
495
|
+
return out
|
|
496
|
+
|
|
497
|
+
# def __hash__(self):
|
|
498
|
+
# return id(self)
|
|
499
|
+
|
|
500
|
+
__hash__ = object.__hash__
|
|
501
|
+
|
|
502
|
+
def __eq__(self, other):
|
|
503
|
+
|
|
504
|
+
if isinstance(other, Tensor):
|
|
505
|
+
other = other.data
|
|
506
|
+
|
|
507
|
+
return self.data == other
|
|
508
|
+
|
|
509
|
+
def gelu(self):
|
|
510
|
+
|
|
511
|
+
x = self.data
|
|
512
|
+
|
|
513
|
+
c = 0.7978845608028654
|
|
514
|
+
|
|
515
|
+
x2 = x * x
|
|
516
|
+
|
|
517
|
+
tanh_inner = np.tanh(c * x * (1.0 + 0.044715 * x2))
|
|
518
|
+
|
|
519
|
+
out = Tensor(
|
|
520
|
+
0.5 * x * (1.0 + tanh_inner),
|
|
521
|
+
requires_grad=self.requires_grad,
|
|
522
|
+
parents=(self,),
|
|
523
|
+
op="GELU",
|
|
524
|
+
)
|
|
525
|
+
|
|
526
|
+
def _backward():
|
|
527
|
+
if not self.requires_grad:
|
|
528
|
+
return
|
|
529
|
+
|
|
530
|
+
sech2 = 1.0 - tanh_inner * tanh_inner
|
|
531
|
+
|
|
532
|
+
grad = 0.5 * (1.0 + tanh_inner) + (0.5 * c) * x * sech2 * (1.0 + (3.0 * 0.044715) * x2)
|
|
533
|
+
|
|
534
|
+
self.grad += out.grad * grad
|
|
535
|
+
|
|
536
|
+
out._backward = _backward
|
|
537
|
+
|
|
538
|
+
return out
|
|
539
|
+
|
|
540
|
+
def sin(self):
|
|
541
|
+
out = Tensor(
|
|
542
|
+
np.sin(self.data),
|
|
543
|
+
requires_grad=self.requires_grad,
|
|
544
|
+
parents=(self, ),
|
|
545
|
+
op="Sin",
|
|
546
|
+
)
|
|
547
|
+
|
|
548
|
+
def _backward():
|
|
549
|
+
# out = sin(self)
|
|
550
|
+
# 𝜹out/𝜹self = cos(self)
|
|
551
|
+
self.grad += out.grad * np.cos(self.data)
|
|
552
|
+
out._backward = _backward
|
|
553
|
+
return out
|
|
554
|
+
|
|
555
|
+
# def mean(self):
|
|
556
|
+
# out = Tensor(
|
|
557
|
+
# self.data.mean(),
|
|
558
|
+
# requires_grad=self.requires_grad,
|
|
559
|
+
# parents=(self,),
|
|
560
|
+
# op="Mean",
|
|
561
|
+
# )
|
|
562
|
+
#
|
|
563
|
+
# def _backward():
|
|
564
|
+
# self.grad += out.grad * np.ones_like(self.data) / self.data.size
|
|
565
|
+
# out._backward = _backward
|
|
566
|
+
# return out
|
|
567
|
+
|
|
568
|
+
def mean(self, axis=None, keepdims=False):
|
|
569
|
+
out = Tensor(
|
|
570
|
+
self.data.mean(axis=axis, keepdims=keepdims),
|
|
571
|
+
requires_grad=self.requires_grad,
|
|
572
|
+
parents=(self,),
|
|
573
|
+
op="Mean",
|
|
574
|
+
)
|
|
575
|
+
|
|
576
|
+
def _backward():
|
|
577
|
+
if not self.requires_grad:
|
|
578
|
+
return
|
|
579
|
+
|
|
580
|
+
grad = out.grad
|
|
581
|
+
|
|
582
|
+
if axis is None:
|
|
583
|
+
count = self.data.size
|
|
584
|
+
else:
|
|
585
|
+
axes = axis if isinstance(axis, tuple) else (axis,)
|
|
586
|
+
count = 1
|
|
587
|
+
for ax in axes:
|
|
588
|
+
count *= self.data.shape[ax]
|
|
589
|
+
|
|
590
|
+
if not keepdims:
|
|
591
|
+
for ax in sorted([a if a >= 0 else a + self.data.ndim for a in axes]):
|
|
592
|
+
grad = np.expand_dims(grad, axis=ax)
|
|
593
|
+
|
|
594
|
+
grad = np.broadcast_to(grad, self.data.shape)
|
|
595
|
+
|
|
596
|
+
self.grad += grad / count
|
|
597
|
+
|
|
598
|
+
out._backward = _backward
|
|
599
|
+
|
|
600
|
+
return out
|
|
601
|
+
|
|
602
|
+
# def sum(self):
|
|
603
|
+
# out = Tensor(
|
|
604
|
+
# self.data.sum(),
|
|
605
|
+
# requires_grad=self.requires_grad,
|
|
606
|
+
# parents=(self,),
|
|
607
|
+
# op="Sum",
|
|
608
|
+
# )
|
|
609
|
+
#
|
|
610
|
+
# def _backward():
|
|
611
|
+
# if self.requires_grad:
|
|
612
|
+
# self.grad += out.grad * np.ones_like(self.data)
|
|
613
|
+
#
|
|
614
|
+
# out._backward = _backward
|
|
615
|
+
# return out
|
|
616
|
+
|
|
617
|
+
def sum(self, axis=None, keepdims=False):
|
|
618
|
+
out = Tensor(
|
|
619
|
+
self.data.sum(axis=axis, keepdims=keepdims),
|
|
620
|
+
requires_grad=self.requires_grad,
|
|
621
|
+
parents=(self,),
|
|
622
|
+
op="Sum",
|
|
623
|
+
)
|
|
624
|
+
|
|
625
|
+
def _backward():
|
|
626
|
+
if not self.requires_grad:
|
|
627
|
+
return
|
|
628
|
+
|
|
629
|
+
grad = out.grad
|
|
630
|
+
|
|
631
|
+
if axis is not None and not keepdims:
|
|
632
|
+
axes = axis if isinstance(axis, tuple) else (axis,)
|
|
633
|
+
for ax in sorted([a if a >= 0 else a + self.data.ndim for a in axes]):
|
|
634
|
+
grad = np.expand_dims(grad, axis=ax)
|
|
635
|
+
|
|
636
|
+
grad = np.broadcast_to(grad, self.data.shape)
|
|
637
|
+
|
|
638
|
+
self.grad += grad
|
|
639
|
+
|
|
640
|
+
out._backward = _backward
|
|
641
|
+
|
|
642
|
+
return out
|
|
643
|
+
|
|
644
|
+
def __truediv__(self, other):
|
|
645
|
+
if not isinstance(other, Tensor):
|
|
646
|
+
other = Tensor(other)
|
|
647
|
+
|
|
648
|
+
out = Tensor(
|
|
649
|
+
self.data / other.data,
|
|
650
|
+
requires_grad=self.requires_grad or other.requires_grad,
|
|
651
|
+
parents=(self, other),
|
|
652
|
+
op="Div",
|
|
653
|
+
)
|
|
654
|
+
|
|
655
|
+
def _backward_():
|
|
656
|
+
if self.requires_grad:
|
|
657
|
+
self.grad += out.grad / other.data
|
|
658
|
+
|
|
659
|
+
if other.requires_grad:
|
|
660
|
+
other.grad -= out.grad * self.data / (other.data ** 2)
|
|
661
|
+
|
|
662
|
+
def _backward():
|
|
663
|
+
|
|
664
|
+
if self.requires_grad:
|
|
665
|
+
self.grad += unbroadcast(
|
|
666
|
+
out.grad / other.data,
|
|
667
|
+
self.shape,
|
|
668
|
+
)
|
|
669
|
+
|
|
670
|
+
if other.requires_grad:
|
|
671
|
+
other.grad += unbroadcast(
|
|
672
|
+
-out.grad * self.data / (other.data ** 2),
|
|
673
|
+
other.shape,
|
|
674
|
+
)
|
|
675
|
+
|
|
676
|
+
out._backward = _backward
|
|
677
|
+
return out
|
|
678
|
+
|
|
679
|
+
def __rtruediv__(self, other):
|
|
680
|
+
return Tensor(other) / self
|
|
681
|
+
|
|
682
|
+
def __radd__(self, other):
|
|
683
|
+
return self + other
|
|
684
|
+
|
|
685
|
+
def __rmul__(self, other):
|
|
686
|
+
return self * other
|
|
687
|
+
|
|
688
|
+
def __rsub__(self, other):
|
|
689
|
+
return Tensor(other) - self
|
|
690
|
+
|
|
691
|
+
def relu(self):
|
|
692
|
+
from src.activations.ReLU import ReLU
|
|
693
|
+
return ReLU.forward(self)
|
|
694
|
+
# out = Tensor(
|
|
695
|
+
# np.maximum(0, self.data),
|
|
696
|
+
# requires_grad=self.requires_grad,
|
|
697
|
+
# parents=(self,),
|
|
698
|
+
# op="ReLU",
|
|
699
|
+
# )
|
|
700
|
+
#
|
|
701
|
+
# def _backward():
|
|
702
|
+
# if self.requires_grad:
|
|
703
|
+
# self.grad += out.grad * (self.data > 0)
|
|
704
|
+
#
|
|
705
|
+
# out._backward = _backward
|
|
706
|
+
#
|
|
707
|
+
# return out
|
|
708
|
+
|
|
709
|
+
def sigmoid(self):
|
|
710
|
+
from src.activations import Sigmoid
|
|
711
|
+
|
|
712
|
+
return Sigmoid.forward(self)
|
|
713
|
+
|
|
714
|
+
def tanh(self):
|
|
715
|
+
from src.activations import Tanh
|
|
716
|
+
return Tanh.forward(self)
|
|
717
|
+
|
|
718
|
+
def log(self):
|
|
719
|
+
from src.math.log import Log
|
|
720
|
+
return Log.forward(self)
|
|
721
|
+
|
|
722
|
+
def exp(self):
|
|
723
|
+
from src.math.exp import Exp
|
|
724
|
+
return Exp.forward(self)
|
|
725
|
+
|
|
726
|
+
# def softmax(self):
|
|
727
|
+
# from src.activations import Softmax
|
|
728
|
+
# return Softmax.forward(self)
|
|
729
|
+
|
|
730
|
+
def softmax(self, axis=-1):
|
|
731
|
+
from src.activations.Softmax import Softmax
|
|
732
|
+
return Softmax.forward(self, axis=axis)
|
|
733
|
+
|
|
734
|
+
def clip(self, min_value, max_value):
|
|
735
|
+
from src.math.clip import Clip
|
|
736
|
+
return Clip.forward(self, min_value, max_value)
|
|
737
|
+
|
|
738
|
+
def argmax(self, axis=None):
|
|
739
|
+
return Tensor(
|
|
740
|
+
np.argmax(self.data, axis=axis),
|
|
741
|
+
requires_grad=False,
|
|
742
|
+
)
|
|
743
|
+
|
|
744
|
+
def item(self):
|
|
745
|
+
return self.data.item()
|
|
746
|
+
|
|
747
|
+
def transpose(self, *axes):
|
|
748
|
+
|
|
749
|
+
out = Tensor(
|
|
750
|
+
np.transpose(self.data, axes),
|
|
751
|
+
requires_grad=self.requires_grad,
|
|
752
|
+
parents=(self,),
|
|
753
|
+
op="Transpose",
|
|
754
|
+
)
|
|
755
|
+
|
|
756
|
+
def _backward():
|
|
757
|
+
if not self.requires_grad:
|
|
758
|
+
return
|
|
759
|
+
|
|
760
|
+
inverse = np.argsort(axes)
|
|
761
|
+
|
|
762
|
+
self.grad += np.transpose(
|
|
763
|
+
out.grad,
|
|
764
|
+
inverse,
|
|
765
|
+
)
|
|
766
|
+
|
|
767
|
+
out._backward = _backward
|
|
768
|
+
|
|
769
|
+
return out
|
|
770
|
+
|
|
771
|
+
def permute(self, *dims):
|
|
772
|
+
return self.transpose(*dims)
|
|
773
|
+
|
|
774
|
+
def reshape(self, *shape):
|
|
775
|
+
|
|
776
|
+
out = Tensor(
|
|
777
|
+
self.data.reshape(shape),
|
|
778
|
+
requires_grad=self.requires_grad,
|
|
779
|
+
parents=(self,),
|
|
780
|
+
op="Reshape",
|
|
781
|
+
)
|
|
782
|
+
|
|
783
|
+
def _backward():
|
|
784
|
+
if self.requires_grad:
|
|
785
|
+
self.grad += out.grad.reshape(
|
|
786
|
+
self.shape
|
|
787
|
+
)
|
|
788
|
+
|
|
789
|
+
out._backward = _backward
|
|
790
|
+
|
|
791
|
+
return out
|
|
792
|
+
|
|
793
|
+
def squeeze(self, axis=None):
|
|
794
|
+
|
|
795
|
+
out = Tensor(
|
|
796
|
+
np.squeeze(self.data, axis),
|
|
797
|
+
requires_grad=self.requires_grad,
|
|
798
|
+
parents=(self,),
|
|
799
|
+
op="Squeeze",
|
|
800
|
+
)
|
|
801
|
+
|
|
802
|
+
def _backward():
|
|
803
|
+
if self.requires_grad:
|
|
804
|
+
self.grad += out.grad.reshape(
|
|
805
|
+
self.shape
|
|
806
|
+
)
|
|
807
|
+
|
|
808
|
+
out._backward = _backward
|
|
809
|
+
|
|
810
|
+
return out
|
|
811
|
+
|
|
812
|
+
def unsqueeze(self, axis):
|
|
813
|
+
|
|
814
|
+
out = Tensor(
|
|
815
|
+
np.expand_dims(self.data, axis),
|
|
816
|
+
requires_grad=self.requires_grad,
|
|
817
|
+
parents=(self,),
|
|
818
|
+
op="Unsqueeze",
|
|
819
|
+
)
|
|
820
|
+
|
|
821
|
+
def _backward():
|
|
822
|
+
if self.requires_grad:
|
|
823
|
+
self.grad += np.squeeze(
|
|
824
|
+
out.grad,
|
|
825
|
+
axis=axis,
|
|
826
|
+
)
|
|
827
|
+
|
|
828
|
+
out._backward = _backward
|
|
829
|
+
|
|
830
|
+
return out
|
|
831
|
+
|
|
832
|
+
def backward(self, grad=None, release=False):
|
|
833
|
+
topo = _topological_order(self)
|
|
834
|
+
|
|
835
|
+
self.grad = np.ones_like(self.data) if grad is None else np.asarray(grad, dtype=np.float32).copy()
|
|
836
|
+
|
|
837
|
+
for t in reversed(topo):
|
|
838
|
+
t._bw()
|
|
839
|
+
|
|
840
|
+
if release and t.parents:
|
|
841
|
+
t._bw = _noop
|
|
842
|
+
t.parents = ()
|
|
843
|
+
t._grad = None
|
|
844
|
+
|
|
845
|
+
@classmethod
|
|
846
|
+
def arange(
|
|
847
|
+
cls,
|
|
848
|
+
start,
|
|
849
|
+
stop=None,
|
|
850
|
+
step=1,
|
|
851
|
+
requires_grad=False,
|
|
852
|
+
):
|
|
853
|
+
if stop is None:
|
|
854
|
+
start, stop = 0, start
|
|
855
|
+
|
|
856
|
+
return cls(
|
|
857
|
+
np.arange(start, stop, step, dtype=np.float32),
|
|
858
|
+
requires_grad=requires_grad,
|
|
859
|
+
)
|
|
860
|
+
|
|
861
|
+
@classmethod
|
|
862
|
+
def random(cls, shape, requires_grad=False):
|
|
863
|
+
return cls(
|
|
864
|
+
np.random.random(shape).astype(np.float32),
|
|
865
|
+
requires_grad=requires_grad,
|
|
866
|
+
)
|