pytensorforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. cli.py +604 -0
  2. pytensorforge-0.1.0.dist-info/METADATA +103 -0
  3. pytensorforge-0.1.0.dist-info/RECORD +146 -0
  4. pytensorforge-0.1.0.dist-info/WHEEL +5 -0
  5. pytensorforge-0.1.0.dist-info/entry_points.txt +2 -0
  6. pytensorforge-0.1.0.dist-info/top_level.txt +2 -0
  7. src/__init__.py +0 -0
  8. src/activations/Activation.py +4 -0
  9. src/activations/ELU.py +11 -0
  10. src/activations/GELU.py +6 -0
  11. src/activations/ReLU.py +27 -0
  12. src/activations/SELU.py +14 -0
  13. src/activations/Sigmoid.py +27 -0
  14. src/activations/Softmax.py +84 -0
  15. src/activations/Tanh.py +29 -0
  16. src/activations/__init__.py +17 -0
  17. src/config.py +120 -0
  18. src/core/Matrix.py +3 -0
  19. src/core/Scalar.py +18 -0
  20. src/core/Tensor.py +866 -0
  21. src/core/Vector.py +31 -0
  22. src/core/__init__.py +0 -0
  23. src/data/__init__.py +0 -0
  24. src/data/chat_dataset.py +188 -0
  25. src/data/corpus.py +104 -0
  26. src/data/document_stream.py +178 -0
  27. src/data/parallel_encode.py +86 -0
  28. src/data/prefetch.py +62 -0
  29. src/data/shard_builder.py +119 -0
  30. src/data/shard_writer.py +81 -0
  31. src/data/sharded_dataset.py +112 -0
  32. src/data/streaming_dataset.py +132 -0
  33. src/data/validation.py +212 -0
  34. src/inference/__init__.py +0 -0
  35. src/inference/chat_template.py +384 -0
  36. src/inference/config.py +48 -0
  37. src/inference/engine.py +241 -0
  38. src/inference/export.py +133 -0
  39. src/inference/kv_cache.py +65 -0
  40. src/inference/runtime.py +161 -0
  41. src/inference/sampling.py +42 -0
  42. src/inference/scheduler.py +473 -0
  43. src/inference/text.py +67 -0
  44. src/initializers/Constant.py +9 -0
  45. src/initializers/GlorotNormal.py +15 -0
  46. src/initializers/GlorotUniform.py +26 -0
  47. src/initializers/HeNormal.py +15 -0
  48. src/initializers/HeUniform.py +14 -0
  49. src/initializers/Initializer.py +4 -0
  50. src/initializers/LecunNormal.py +16 -0
  51. src/initializers/LecunUniform.py +14 -0
  52. src/initializers/Ones.py +6 -0
  53. src/initializers/Orthogonal.py +14 -0
  54. src/initializers/RandomNormal.py +14 -0
  55. src/initializers/RandomUniform.py +14 -0
  56. src/initializers/Zeros.py +8 -0
  57. src/initializers/__init__.py +17 -0
  58. src/loss/CategoricalCrossEntropy.py +9 -0
  59. src/loss/CrossEntropyLoss.py +34 -0
  60. src/loss/CrossEntropyWithLogitsLoss.py +59 -0
  61. src/loss/Hinge.py +5 -0
  62. src/loss/Huber.py +22 -0
  63. src/loss/Loss.py +6 -0
  64. src/loss/MSE.py +7 -0
  65. src/loss/MSELoss.py +10 -0
  66. src/loss/SparseCategoricalCrossEntropy.py +15 -0
  67. src/loss/__init__.py +18 -0
  68. src/loss/bce.py +34 -0
  69. src/loss/mae.py +16 -0
  70. src/math/__init__.py +0 -0
  71. src/math/clip.py +37 -0
  72. src/math/exp.py +27 -0
  73. src/math/log.py +25 -0
  74. src/math/sigmoid.py +5 -0
  75. src/models/__init__.py +0 -0
  76. src/models/embedding/Embedding.py +65 -0
  77. src/models/embedding/__init__.py +0 -0
  78. src/models/gpt/__init__.py +0 -0
  79. src/models/gpt/attention.py +158 -0
  80. src/models/gpt/block.py +74 -0
  81. src/models/gpt/config.py +103 -0
  82. src/models/gpt/context.py +44 -0
  83. src/models/gpt/model.py +165 -0
  84. src/models/gpt/recompute.py +35 -0
  85. src/models/gpt/rope.py +84 -0
  86. src/models/regression/Linear.py +51 -0
  87. src/models/regression/Logistic.py +36 -0
  88. src/models/regression/__init__.py +0 -0
  89. src/models/seq/Sequential.py +297 -0
  90. src/models/seq/__init__.py +0 -0
  91. src/models/svm/__init__.py +0 -0
  92. src/models/tokenizer/BPETokenizer.py +228 -0
  93. src/models/tokenizer/__init__.py +0 -0
  94. src/models/transformers/Dropout.py +35 -0
  95. src/models/transformers/LastToken.py +10 -0
  96. src/models/transformers/LayerNorm.py +54 -0
  97. src/models/transformers/Linear.py +18 -0
  98. src/models/transformers/MultiHeadAttention.py +130 -0
  99. src/models/transformers/TransformerBlock.py +79 -0
  100. src/models/transformers/__init__.py +0 -0
  101. src/neural/Dense.py +58 -0
  102. src/neural/LSTM.py +167 -0
  103. src/neural/Layer.py +72 -0
  104. src/neural/Parameter.py +30 -0
  105. src/neural/RNN.py +83 -0
  106. src/neural/__init__.py +0 -0
  107. src/ops/__init__.py +0 -0
  108. src/ops/stack.py +40 -0
  109. src/optimizers/Adagrad.py +31 -0
  110. src/optimizers/Adam.py +98 -0
  111. src/optimizers/AdamW.py +84 -0
  112. src/optimizers/Batch.py +11 -0
  113. src/optimizers/Nesterov.py +35 -0
  114. src/optimizers/Optimizer.py +18 -0
  115. src/optimizers/RMSProp.py +35 -0
  116. src/optimizers/SGD.py +30 -0
  117. src/optimizers/SGDMomentum.py +28 -0
  118. src/optimizers/__init__.py +9 -0
  119. src/scaling/StandardScaler.py +15 -0
  120. src/scaling/__init__.py +0 -0
  121. src/serialization/__init__.py +0 -0
  122. src/serialization/checkpoint.py +58 -0
  123. src/serialization/modelio.py +132 -0
  124. src/serving/__init__.py +0 -0
  125. src/serving/app.py +792 -0
  126. src/serving/config.py +216 -0
  127. src/serving/errors.py +51 -0
  128. src/serving/http.py +599 -0
  129. src/serving/metrics.py +293 -0
  130. src/serving/model_server.py +287 -0
  131. src/serving/protocol.py +377 -0
  132. src/serving/security.py +200 -0
  133. src/serving/server.py +121 -0
  134. src/tokenization/__init__.py +0 -0
  135. src/tokenization/base.py +75 -0
  136. src/tokenization/bpe.py +190 -0
  137. src/tokenization/bytebpe.py +476 -0
  138. src/tokenization/registry.py +28 -0
  139. src/training/__init__.py +0 -0
  140. src/training/checkpoint_manager.py +101 -0
  141. src/training/experiment.py +71 -0
  142. src/training/losses.py +42 -0
  143. src/training/precision.py +141 -0
  144. src/training/profiler.py +38 -0
  145. src/training/scheduler.py +50 -0
  146. src/training/trainer.py +594 -0
src/core/Tensor.py ADDED
@@ -0,0 +1,866 @@
1
+ import threading
2
+
3
+ import numpy as np
4
+
5
+ _state = threading.local()
6
+
7
+
8
+ def _noop():
9
+ return None
10
+
11
+
12
+ def grad_enabled():
13
+ return getattr(_state, "grad_enabled", True)
14
+
15
+
16
+ def matmul_policy():
17
+ return getattr(_state, "matmul_policy", None)
18
+
19
+
20
+ def set_matmul_policy(fn):
21
+ previous = matmul_policy()
22
+ _state.matmul_policy = fn
23
+ return previous
24
+
25
+
26
+ class no_grad:
27
+
28
+ def __enter__(self):
29
+ self._previous = grad_enabled()
30
+ _state.grad_enabled = False
31
+ return self
32
+
33
+ def __exit__(self, *exc):
34
+ _state.grad_enabled = self._previous
35
+ return False
36
+
37
+
38
+ def _round_operand(x):
39
+ policy = matmul_policy()
40
+ return x if policy is None else policy(x)
41
+
42
+ # x = Tensor([[1, 2],
43
+ # [3, 4]], requires_grad=True)
44
+ #
45
+ # print(x.shape)
46
+
47
+ def unbroadcast(grad, shape):
48
+ while grad.ndim > len(shape):
49
+ grad = grad.sum(axis=0)
50
+
51
+ for axis, size in enumerate(shape):
52
+ if size == 1:
53
+ grad = grad.sum(axis=axis, keepdims=True)
54
+
55
+ return grad
56
+
57
+ def _topological_order(root):
58
+ order = []
59
+ visited = {id(root)}
60
+ stack = [(root, iter(root.parents))]
61
+
62
+ while stack:
63
+ node, children = stack[-1]
64
+ advanced = False
65
+
66
+ for child in children:
67
+ if id(child) not in visited:
68
+ visited.add(id(child))
69
+ stack.append((child, iter(child.parents)))
70
+ advanced = True
71
+ break
72
+
73
+ if not advanced:
74
+ stack.pop()
75
+ order.append(node)
76
+
77
+ return order
78
+
79
+
80
+ class Tensor:
81
+
82
+ def __init__(
83
+ self,
84
+ data,
85
+ requires_grad=True,
86
+ parents=(),
87
+ op=None,
88
+ ):
89
+ self.data = np.asarray(data, dtype=np.float32)
90
+
91
+ self._grad = None
92
+
93
+ if not grad_enabled():
94
+ requires_grad = False
95
+ parents = ()
96
+
97
+ self.requires_grad = requires_grad
98
+
99
+ self.parents = parents
100
+
101
+ self.op = op
102
+
103
+ self._bw = _noop
104
+
105
+ @property
106
+ def grad(self):
107
+ if self._grad is None:
108
+ self._grad = np.zeros_like(self.data)
109
+ return self._grad
110
+
111
+ @grad.setter
112
+ def grad(self, value):
113
+ self._grad = value
114
+
115
+ @property
116
+ def _backward(self):
117
+ return self._bw
118
+
119
+ @_backward.setter
120
+ def _backward(self, fn):
121
+ self._bw = fn if self.requires_grad else _noop
122
+
123
+ @property
124
+ def shape(self):
125
+ return self.data.shape
126
+
127
+ def zero_grad(self):
128
+ if self._grad is not None:
129
+ self._grad.fill(0)
130
+
131
+ def __len__(self):
132
+ return self.data.shape[0]
133
+
134
+ # def __getitem__(self, idx):
135
+ # return Tensor(
136
+ # self.data[idx],
137
+ # requires_grad=self.requires_grad,
138
+ # )
139
+
140
+ def __getitem__(self, idx):
141
+
142
+ if isinstance(idx, Tensor):
143
+ idx = idx.data.astype(np.int64)
144
+
145
+ elif isinstance(idx, tuple):
146
+ idx = tuple(
147
+ i.data.astype(np.int64) if isinstance(i, Tensor) else i
148
+ for i in idx
149
+ )
150
+
151
+ out = Tensor(
152
+ self.data[idx],
153
+ requires_grad=self.requires_grad,
154
+ parents=(self,),
155
+ op="Slice",
156
+ )
157
+
158
+ def _backward():
159
+ if not self.requires_grad:
160
+ return
161
+
162
+ np.add.at(
163
+ self.grad,
164
+ idx,
165
+ out.grad,
166
+ )
167
+
168
+ out._backward = _backward
169
+
170
+ return out
171
+
172
+ @staticmethod
173
+ def zeros(shape, requires_grad=False):
174
+ return Tensor(
175
+ np.zeros(shape, dtype=np.float32),
176
+ requires_grad=requires_grad,
177
+ )
178
+
179
+ @staticmethod
180
+ def stack(tensors, axis=0):
181
+ from src.ops.stack import Stack
182
+ return Stack.forward(tensors, axis)
183
+
184
+ def __repr__(self):
185
+ return f"Tensor(data={self.data}, requires_grad={self.requires_grad})"
186
+
187
+ def __add__(self, other):
188
+ if not isinstance(other, Tensor):
189
+ other = Tensor(other)
190
+
191
+ out = Tensor(
192
+ self.data + other.data,
193
+ requires_grad=self.requires_grad or other.requires_grad,
194
+ parents=(self, other),
195
+ op="Add",
196
+ )
197
+
198
+ def backward():
199
+ if self.requires_grad:
200
+ self.grad += unbroadcast(out.grad, self.shape)
201
+
202
+ if other.requires_grad:
203
+ other.grad += unbroadcast(out.grad, other.shape)
204
+
205
+ out._backward = backward
206
+
207
+ return out
208
+
209
+ def __sub__(self, other):
210
+
211
+ if not isinstance(other, Tensor):
212
+ other = Tensor(other)
213
+
214
+ out = Tensor(
215
+ self.data - other.data,
216
+ requires_grad=self.requires_grad or other.requires_grad,
217
+ parents=(self, other),
218
+ op="Sub",
219
+ )
220
+
221
+ def _backward_():
222
+ # out = self - other
223
+ # 𝜹out/𝜹self = 1 - 0 = 1
224
+ self.grad += out.grad * 1
225
+ # 𝜹out/𝜹other = 0 - 1 = -1
226
+ other.grad += out.grad * (-1)
227
+
228
+ def _backward():
229
+ if self.requires_grad:
230
+ self.grad += unbroadcast(out.grad, self.shape)
231
+
232
+ if other.requires_grad:
233
+ other.grad += unbroadcast(-out.grad, other.shape)
234
+
235
+ out._backward = _backward
236
+ return out
237
+
238
+ # def __neg__(self):
239
+ # return self * -1
240
+
241
+ def __neg__(self):
242
+ out = Tensor(
243
+ -self.data,
244
+ requires_grad=self.requires_grad,
245
+ parents=(self,),
246
+ op="Neg",
247
+ )
248
+
249
+ def _backward():
250
+ if self.requires_grad:
251
+ self.grad -= out.grad
252
+
253
+ out._backward = _backward
254
+
255
+ return out
256
+
257
+ def __mul__(self, other):
258
+ if not isinstance(other, Tensor):
259
+ other = Tensor(other)
260
+
261
+ out = Tensor(
262
+ self.data * other.data,
263
+ requires_grad=self.requires_grad or other.requires_grad,
264
+ parents=(self, other),
265
+ op="Mul",
266
+ )
267
+
268
+ def _backward_():
269
+ # out = self * other
270
+ # 𝜹out/𝜹self = other
271
+ self.grad += out.grad * other.data
272
+ # out = self * other
273
+ # 𝜹out/𝜹other = self
274
+ other.grad += out.grad * self.data
275
+
276
+ def _backward():
277
+ if self.requires_grad:
278
+ self.grad += unbroadcast(
279
+ out.grad * other.data,
280
+ self.shape,
281
+ )
282
+
283
+ if other.requires_grad:
284
+ other.grad += unbroadcast(
285
+ out.grad * self.data,
286
+ other.shape,
287
+ )
288
+
289
+ out._backward = _backward
290
+ return out
291
+
292
+ # def __matmul__(self, other):
293
+ # if not isinstance(other, Tensor):
294
+ # other = Tensor(other)
295
+ #
296
+ # out = Tensor(self.data @ other.data, requires_grad=self.requires_grad or other.requires_grad, parents=(self, other), op="MatMul")
297
+ # def _backward():
298
+ # self.grad += out.grad @ other.data.T
299
+ # other.grad += self.data.T @ out.grad
300
+ # out._backward = _backward
301
+ # return out
302
+
303
+ def __matmul__(self, other):
304
+
305
+ if not isinstance(other, Tensor):
306
+ other = Tensor(other)
307
+
308
+ policy = matmul_policy()
309
+ a = self.data if policy is None else policy(self.data)
310
+ b = other.data if policy is None else policy(other.data)
311
+ result = a @ b
312
+
313
+ out = Tensor(
314
+ result if policy is None else policy(result),
315
+ requires_grad=self.requires_grad or other.requires_grad,
316
+ parents=(self, other),
317
+ op="MatMul",
318
+ )
319
+
320
+ def _backward():
321
+ g = out.grad if policy is None else policy(out.grad)
322
+
323
+ if self.requires_grad:
324
+ grad = np.matmul(
325
+ g,
326
+ np.swapaxes(b, -1, -2),
327
+ )
328
+
329
+ if policy is not None:
330
+ grad = policy(grad)
331
+
332
+ self.grad += Tensor.unbroadcast(
333
+ grad,
334
+ self.shape,
335
+ )
336
+
337
+ if other.requires_grad:
338
+ grad = np.matmul(
339
+ np.swapaxes(a, -1, -2),
340
+ g,
341
+ )
342
+
343
+ if policy is not None:
344
+ grad = policy(grad)
345
+
346
+ other.grad += Tensor.unbroadcast(
347
+ grad,
348
+ other.shape,
349
+ )
350
+
351
+ out._backward = _backward
352
+
353
+ return out
354
+
355
+ @staticmethod
356
+ def unbroadcast(grad, shape):
357
+ while grad.ndim > len(shape):
358
+ grad = grad.sum(axis=0)
359
+
360
+ for axis, size in enumerate(shape):
361
+ if size == 1:
362
+ grad = grad.sum(axis=axis, keepdims=True)
363
+
364
+ return grad
365
+
366
+ def __div__(self, other):
367
+ if not isinstance(other, Tensor):
368
+ other = Tensor(other)
369
+
370
+ out = Tensor(
371
+ self.data / other.data,
372
+ requires_grad=self.requires_grad or other.requires_grad,
373
+ parents=(self, other),
374
+ op="Div",
375
+ )
376
+
377
+ def _backward():
378
+ # out = self / other
379
+ # 𝜹out/𝜹self = 1/other
380
+ # 𝜹out/𝜹other = self * (-1) * other ** (-2)
381
+ self.grad += out.grad * ( 1 / other.data)
382
+ other.grad += out.grad * (-self.data / (other.data ** 2))
383
+ out._backward = _backward
384
+ return out
385
+
386
+ # def __pow__(self, other):
387
+ # if not isinstance(other, Tensor):
388
+ # other = Tensor(other)
389
+ #
390
+ # out = Tensor(
391
+ # self.data ** other,
392
+ # requires_grad=self.requires_grad,
393
+ # parents=(self, ),
394
+ # op="Pow",
395
+ # )
396
+ #
397
+ # def _backward():
398
+ # # out = self ** other
399
+ # # 𝜹out/𝜹self = other * self ** (other - 1)
400
+ # self.grad += out.grad * (other * self.data ** (other - 1))
401
+ # out._backward = _backward
402
+ # return out
403
+
404
+ def __pow__(self, other):
405
+
406
+ if isinstance(other, Tensor):
407
+ other = other.data
408
+
409
+ out = Tensor(
410
+ self.data ** other,
411
+ requires_grad=self.requires_grad,
412
+ parents=(self,),
413
+ op="Pow",
414
+ )
415
+
416
+ def _backward():
417
+ if self.requires_grad:
418
+ self.grad += out.grad * other * (self.data ** (other - 1))
419
+
420
+ out._backward = _backward
421
+
422
+ return out
423
+
424
+ def __rpow__(self, other):
425
+
426
+ out = Tensor(
427
+ other ** self.data,
428
+ requires_grad=self.requires_grad,
429
+ parents=(self,),
430
+ op="RPow",
431
+ )
432
+
433
+ def _backward():
434
+ if self.requires_grad:
435
+ self.grad += (
436
+ out.grad
437
+ * np.log(other)
438
+ * (other ** self.data)
439
+ )
440
+
441
+ out._backward = _backward
442
+
443
+ return out
444
+
445
+ def sqrt(self):
446
+
447
+ out = Tensor(
448
+ np.sqrt(self.data),
449
+ requires_grad=self.requires_grad,
450
+ parents=(self,),
451
+ op="Sqrt",
452
+ )
453
+
454
+ def _backward():
455
+ if not self.requires_grad:
456
+ return
457
+
458
+ self.grad += out.grad * (0.5 / np.sqrt(self.data))
459
+
460
+ out._backward = _backward
461
+
462
+ return out
463
+
464
+ def masked_fill(self, mask, value):
465
+
466
+ if isinstance(mask, Tensor):
467
+ mask = mask.data
468
+
469
+ mask = np.broadcast_to(mask, self.data.shape)
470
+
471
+ out_data = self.data.copy()
472
+
473
+ out_data[mask] = value
474
+
475
+ out = Tensor(
476
+ out_data,
477
+ requires_grad=self.requires_grad,
478
+ parents=(self,),
479
+ op="MaskedFill",
480
+ )
481
+
482
+ def _backward():
483
+
484
+ if not self.requires_grad:
485
+ return
486
+
487
+ grad = out.grad.copy()
488
+
489
+ grad[mask] = 0
490
+
491
+ self.grad += grad
492
+
493
+ out._backward = _backward
494
+
495
+ return out
496
+
497
+ # def __hash__(self):
498
+ # return id(self)
499
+
500
+ __hash__ = object.__hash__
501
+
502
+ def __eq__(self, other):
503
+
504
+ if isinstance(other, Tensor):
505
+ other = other.data
506
+
507
+ return self.data == other
508
+
509
+ def gelu(self):
510
+
511
+ x = self.data
512
+
513
+ c = 0.7978845608028654
514
+
515
+ x2 = x * x
516
+
517
+ tanh_inner = np.tanh(c * x * (1.0 + 0.044715 * x2))
518
+
519
+ out = Tensor(
520
+ 0.5 * x * (1.0 + tanh_inner),
521
+ requires_grad=self.requires_grad,
522
+ parents=(self,),
523
+ op="GELU",
524
+ )
525
+
526
+ def _backward():
527
+ if not self.requires_grad:
528
+ return
529
+
530
+ sech2 = 1.0 - tanh_inner * tanh_inner
531
+
532
+ grad = 0.5 * (1.0 + tanh_inner) + (0.5 * c) * x * sech2 * (1.0 + (3.0 * 0.044715) * x2)
533
+
534
+ self.grad += out.grad * grad
535
+
536
+ out._backward = _backward
537
+
538
+ return out
539
+
540
+ def sin(self):
541
+ out = Tensor(
542
+ np.sin(self.data),
543
+ requires_grad=self.requires_grad,
544
+ parents=(self, ),
545
+ op="Sin",
546
+ )
547
+
548
+ def _backward():
549
+ # out = sin(self)
550
+ # 𝜹out/𝜹self = cos(self)
551
+ self.grad += out.grad * np.cos(self.data)
552
+ out._backward = _backward
553
+ return out
554
+
555
+ # def mean(self):
556
+ # out = Tensor(
557
+ # self.data.mean(),
558
+ # requires_grad=self.requires_grad,
559
+ # parents=(self,),
560
+ # op="Mean",
561
+ # )
562
+ #
563
+ # def _backward():
564
+ # self.grad += out.grad * np.ones_like(self.data) / self.data.size
565
+ # out._backward = _backward
566
+ # return out
567
+
568
+ def mean(self, axis=None, keepdims=False):
569
+ out = Tensor(
570
+ self.data.mean(axis=axis, keepdims=keepdims),
571
+ requires_grad=self.requires_grad,
572
+ parents=(self,),
573
+ op="Mean",
574
+ )
575
+
576
+ def _backward():
577
+ if not self.requires_grad:
578
+ return
579
+
580
+ grad = out.grad
581
+
582
+ if axis is None:
583
+ count = self.data.size
584
+ else:
585
+ axes = axis if isinstance(axis, tuple) else (axis,)
586
+ count = 1
587
+ for ax in axes:
588
+ count *= self.data.shape[ax]
589
+
590
+ if not keepdims:
591
+ for ax in sorted([a if a >= 0 else a + self.data.ndim for a in axes]):
592
+ grad = np.expand_dims(grad, axis=ax)
593
+
594
+ grad = np.broadcast_to(grad, self.data.shape)
595
+
596
+ self.grad += grad / count
597
+
598
+ out._backward = _backward
599
+
600
+ return out
601
+
602
+ # def sum(self):
603
+ # out = Tensor(
604
+ # self.data.sum(),
605
+ # requires_grad=self.requires_grad,
606
+ # parents=(self,),
607
+ # op="Sum",
608
+ # )
609
+ #
610
+ # def _backward():
611
+ # if self.requires_grad:
612
+ # self.grad += out.grad * np.ones_like(self.data)
613
+ #
614
+ # out._backward = _backward
615
+ # return out
616
+
617
+ def sum(self, axis=None, keepdims=False):
618
+ out = Tensor(
619
+ self.data.sum(axis=axis, keepdims=keepdims),
620
+ requires_grad=self.requires_grad,
621
+ parents=(self,),
622
+ op="Sum",
623
+ )
624
+
625
+ def _backward():
626
+ if not self.requires_grad:
627
+ return
628
+
629
+ grad = out.grad
630
+
631
+ if axis is not None and not keepdims:
632
+ axes = axis if isinstance(axis, tuple) else (axis,)
633
+ for ax in sorted([a if a >= 0 else a + self.data.ndim for a in axes]):
634
+ grad = np.expand_dims(grad, axis=ax)
635
+
636
+ grad = np.broadcast_to(grad, self.data.shape)
637
+
638
+ self.grad += grad
639
+
640
+ out._backward = _backward
641
+
642
+ return out
643
+
644
+ def __truediv__(self, other):
645
+ if not isinstance(other, Tensor):
646
+ other = Tensor(other)
647
+
648
+ out = Tensor(
649
+ self.data / other.data,
650
+ requires_grad=self.requires_grad or other.requires_grad,
651
+ parents=(self, other),
652
+ op="Div",
653
+ )
654
+
655
+ def _backward_():
656
+ if self.requires_grad:
657
+ self.grad += out.grad / other.data
658
+
659
+ if other.requires_grad:
660
+ other.grad -= out.grad * self.data / (other.data ** 2)
661
+
662
+ def _backward():
663
+
664
+ if self.requires_grad:
665
+ self.grad += unbroadcast(
666
+ out.grad / other.data,
667
+ self.shape,
668
+ )
669
+
670
+ if other.requires_grad:
671
+ other.grad += unbroadcast(
672
+ -out.grad * self.data / (other.data ** 2),
673
+ other.shape,
674
+ )
675
+
676
+ out._backward = _backward
677
+ return out
678
+
679
+ def __rtruediv__(self, other):
680
+ return Tensor(other) / self
681
+
682
+ def __radd__(self, other):
683
+ return self + other
684
+
685
+ def __rmul__(self, other):
686
+ return self * other
687
+
688
+ def __rsub__(self, other):
689
+ return Tensor(other) - self
690
+
691
+ def relu(self):
692
+ from src.activations.ReLU import ReLU
693
+ return ReLU.forward(self)
694
+ # out = Tensor(
695
+ # np.maximum(0, self.data),
696
+ # requires_grad=self.requires_grad,
697
+ # parents=(self,),
698
+ # op="ReLU",
699
+ # )
700
+ #
701
+ # def _backward():
702
+ # if self.requires_grad:
703
+ # self.grad += out.grad * (self.data > 0)
704
+ #
705
+ # out._backward = _backward
706
+ #
707
+ # return out
708
+
709
+ def sigmoid(self):
710
+ from src.activations import Sigmoid
711
+
712
+ return Sigmoid.forward(self)
713
+
714
+ def tanh(self):
715
+ from src.activations import Tanh
716
+ return Tanh.forward(self)
717
+
718
+ def log(self):
719
+ from src.math.log import Log
720
+ return Log.forward(self)
721
+
722
+ def exp(self):
723
+ from src.math.exp import Exp
724
+ return Exp.forward(self)
725
+
726
+ # def softmax(self):
727
+ # from src.activations import Softmax
728
+ # return Softmax.forward(self)
729
+
730
+ def softmax(self, axis=-1):
731
+ from src.activations.Softmax import Softmax
732
+ return Softmax.forward(self, axis=axis)
733
+
734
+ def clip(self, min_value, max_value):
735
+ from src.math.clip import Clip
736
+ return Clip.forward(self, min_value, max_value)
737
+
738
+ def argmax(self, axis=None):
739
+ return Tensor(
740
+ np.argmax(self.data, axis=axis),
741
+ requires_grad=False,
742
+ )
743
+
744
+ def item(self):
745
+ return self.data.item()
746
+
747
+ def transpose(self, *axes):
748
+
749
+ out = Tensor(
750
+ np.transpose(self.data, axes),
751
+ requires_grad=self.requires_grad,
752
+ parents=(self,),
753
+ op="Transpose",
754
+ )
755
+
756
+ def _backward():
757
+ if not self.requires_grad:
758
+ return
759
+
760
+ inverse = np.argsort(axes)
761
+
762
+ self.grad += np.transpose(
763
+ out.grad,
764
+ inverse,
765
+ )
766
+
767
+ out._backward = _backward
768
+
769
+ return out
770
+
771
+ def permute(self, *dims):
772
+ return self.transpose(*dims)
773
+
774
+ def reshape(self, *shape):
775
+
776
+ out = Tensor(
777
+ self.data.reshape(shape),
778
+ requires_grad=self.requires_grad,
779
+ parents=(self,),
780
+ op="Reshape",
781
+ )
782
+
783
+ def _backward():
784
+ if self.requires_grad:
785
+ self.grad += out.grad.reshape(
786
+ self.shape
787
+ )
788
+
789
+ out._backward = _backward
790
+
791
+ return out
792
+
793
+ def squeeze(self, axis=None):
794
+
795
+ out = Tensor(
796
+ np.squeeze(self.data, axis),
797
+ requires_grad=self.requires_grad,
798
+ parents=(self,),
799
+ op="Squeeze",
800
+ )
801
+
802
+ def _backward():
803
+ if self.requires_grad:
804
+ self.grad += out.grad.reshape(
805
+ self.shape
806
+ )
807
+
808
+ out._backward = _backward
809
+
810
+ return out
811
+
812
+ def unsqueeze(self, axis):
813
+
814
+ out = Tensor(
815
+ np.expand_dims(self.data, axis),
816
+ requires_grad=self.requires_grad,
817
+ parents=(self,),
818
+ op="Unsqueeze",
819
+ )
820
+
821
+ def _backward():
822
+ if self.requires_grad:
823
+ self.grad += np.squeeze(
824
+ out.grad,
825
+ axis=axis,
826
+ )
827
+
828
+ out._backward = _backward
829
+
830
+ return out
831
+
832
+ def backward(self, grad=None, release=False):
833
+ topo = _topological_order(self)
834
+
835
+ self.grad = np.ones_like(self.data) if grad is None else np.asarray(grad, dtype=np.float32).copy()
836
+
837
+ for t in reversed(topo):
838
+ t._bw()
839
+
840
+ if release and t.parents:
841
+ t._bw = _noop
842
+ t.parents = ()
843
+ t._grad = None
844
+
845
+ @classmethod
846
+ def arange(
847
+ cls,
848
+ start,
849
+ stop=None,
850
+ step=1,
851
+ requires_grad=False,
852
+ ):
853
+ if stop is None:
854
+ start, stop = 0, start
855
+
856
+ return cls(
857
+ np.arange(start, stop, step, dtype=np.float32),
858
+ requires_grad=requires_grad,
859
+ )
860
+
861
+ @classmethod
862
+ def random(cls, shape, requires_grad=False):
863
+ return cls(
864
+ np.random.random(shape).astype(np.float32),
865
+ requires_grad=requires_grad,
866
+ )