pytensorforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. cli.py +604 -0
  2. pytensorforge-0.1.0.dist-info/METADATA +103 -0
  3. pytensorforge-0.1.0.dist-info/RECORD +146 -0
  4. pytensorforge-0.1.0.dist-info/WHEEL +5 -0
  5. pytensorforge-0.1.0.dist-info/entry_points.txt +2 -0
  6. pytensorforge-0.1.0.dist-info/top_level.txt +2 -0
  7. src/__init__.py +0 -0
  8. src/activations/Activation.py +4 -0
  9. src/activations/ELU.py +11 -0
  10. src/activations/GELU.py +6 -0
  11. src/activations/ReLU.py +27 -0
  12. src/activations/SELU.py +14 -0
  13. src/activations/Sigmoid.py +27 -0
  14. src/activations/Softmax.py +84 -0
  15. src/activations/Tanh.py +29 -0
  16. src/activations/__init__.py +17 -0
  17. src/config.py +120 -0
  18. src/core/Matrix.py +3 -0
  19. src/core/Scalar.py +18 -0
  20. src/core/Tensor.py +866 -0
  21. src/core/Vector.py +31 -0
  22. src/core/__init__.py +0 -0
  23. src/data/__init__.py +0 -0
  24. src/data/chat_dataset.py +188 -0
  25. src/data/corpus.py +104 -0
  26. src/data/document_stream.py +178 -0
  27. src/data/parallel_encode.py +86 -0
  28. src/data/prefetch.py +62 -0
  29. src/data/shard_builder.py +119 -0
  30. src/data/shard_writer.py +81 -0
  31. src/data/sharded_dataset.py +112 -0
  32. src/data/streaming_dataset.py +132 -0
  33. src/data/validation.py +212 -0
  34. src/inference/__init__.py +0 -0
  35. src/inference/chat_template.py +384 -0
  36. src/inference/config.py +48 -0
  37. src/inference/engine.py +241 -0
  38. src/inference/export.py +133 -0
  39. src/inference/kv_cache.py +65 -0
  40. src/inference/runtime.py +161 -0
  41. src/inference/sampling.py +42 -0
  42. src/inference/scheduler.py +473 -0
  43. src/inference/text.py +67 -0
  44. src/initializers/Constant.py +9 -0
  45. src/initializers/GlorotNormal.py +15 -0
  46. src/initializers/GlorotUniform.py +26 -0
  47. src/initializers/HeNormal.py +15 -0
  48. src/initializers/HeUniform.py +14 -0
  49. src/initializers/Initializer.py +4 -0
  50. src/initializers/LecunNormal.py +16 -0
  51. src/initializers/LecunUniform.py +14 -0
  52. src/initializers/Ones.py +6 -0
  53. src/initializers/Orthogonal.py +14 -0
  54. src/initializers/RandomNormal.py +14 -0
  55. src/initializers/RandomUniform.py +14 -0
  56. src/initializers/Zeros.py +8 -0
  57. src/initializers/__init__.py +17 -0
  58. src/loss/CategoricalCrossEntropy.py +9 -0
  59. src/loss/CrossEntropyLoss.py +34 -0
  60. src/loss/CrossEntropyWithLogitsLoss.py +59 -0
  61. src/loss/Hinge.py +5 -0
  62. src/loss/Huber.py +22 -0
  63. src/loss/Loss.py +6 -0
  64. src/loss/MSE.py +7 -0
  65. src/loss/MSELoss.py +10 -0
  66. src/loss/SparseCategoricalCrossEntropy.py +15 -0
  67. src/loss/__init__.py +18 -0
  68. src/loss/bce.py +34 -0
  69. src/loss/mae.py +16 -0
  70. src/math/__init__.py +0 -0
  71. src/math/clip.py +37 -0
  72. src/math/exp.py +27 -0
  73. src/math/log.py +25 -0
  74. src/math/sigmoid.py +5 -0
  75. src/models/__init__.py +0 -0
  76. src/models/embedding/Embedding.py +65 -0
  77. src/models/embedding/__init__.py +0 -0
  78. src/models/gpt/__init__.py +0 -0
  79. src/models/gpt/attention.py +158 -0
  80. src/models/gpt/block.py +74 -0
  81. src/models/gpt/config.py +103 -0
  82. src/models/gpt/context.py +44 -0
  83. src/models/gpt/model.py +165 -0
  84. src/models/gpt/recompute.py +35 -0
  85. src/models/gpt/rope.py +84 -0
  86. src/models/regression/Linear.py +51 -0
  87. src/models/regression/Logistic.py +36 -0
  88. src/models/regression/__init__.py +0 -0
  89. src/models/seq/Sequential.py +297 -0
  90. src/models/seq/__init__.py +0 -0
  91. src/models/svm/__init__.py +0 -0
  92. src/models/tokenizer/BPETokenizer.py +228 -0
  93. src/models/tokenizer/__init__.py +0 -0
  94. src/models/transformers/Dropout.py +35 -0
  95. src/models/transformers/LastToken.py +10 -0
  96. src/models/transformers/LayerNorm.py +54 -0
  97. src/models/transformers/Linear.py +18 -0
  98. src/models/transformers/MultiHeadAttention.py +130 -0
  99. src/models/transformers/TransformerBlock.py +79 -0
  100. src/models/transformers/__init__.py +0 -0
  101. src/neural/Dense.py +58 -0
  102. src/neural/LSTM.py +167 -0
  103. src/neural/Layer.py +72 -0
  104. src/neural/Parameter.py +30 -0
  105. src/neural/RNN.py +83 -0
  106. src/neural/__init__.py +0 -0
  107. src/ops/__init__.py +0 -0
  108. src/ops/stack.py +40 -0
  109. src/optimizers/Adagrad.py +31 -0
  110. src/optimizers/Adam.py +98 -0
  111. src/optimizers/AdamW.py +84 -0
  112. src/optimizers/Batch.py +11 -0
  113. src/optimizers/Nesterov.py +35 -0
  114. src/optimizers/Optimizer.py +18 -0
  115. src/optimizers/RMSProp.py +35 -0
  116. src/optimizers/SGD.py +30 -0
  117. src/optimizers/SGDMomentum.py +28 -0
  118. src/optimizers/__init__.py +9 -0
  119. src/scaling/StandardScaler.py +15 -0
  120. src/scaling/__init__.py +0 -0
  121. src/serialization/__init__.py +0 -0
  122. src/serialization/checkpoint.py +58 -0
  123. src/serialization/modelio.py +132 -0
  124. src/serving/__init__.py +0 -0
  125. src/serving/app.py +792 -0
  126. src/serving/config.py +216 -0
  127. src/serving/errors.py +51 -0
  128. src/serving/http.py +599 -0
  129. src/serving/metrics.py +293 -0
  130. src/serving/model_server.py +287 -0
  131. src/serving/protocol.py +377 -0
  132. src/serving/security.py +200 -0
  133. src/serving/server.py +121 -0
  134. src/tokenization/__init__.py +0 -0
  135. src/tokenization/base.py +75 -0
  136. src/tokenization/bpe.py +190 -0
  137. src/tokenization/bytebpe.py +476 -0
  138. src/tokenization/registry.py +28 -0
  139. src/training/__init__.py +0 -0
  140. src/training/checkpoint_manager.py +101 -0
  141. src/training/experiment.py +71 -0
  142. src/training/losses.py +42 -0
  143. src/training/precision.py +141 -0
  144. src/training/profiler.py +38 -0
  145. src/training/scheduler.py +50 -0
  146. src/training/trainer.py +594 -0
@@ -0,0 +1,130 @@
1
+ from src.core.Tensor import Tensor
2
+ from src.neural.Layer import Layer
3
+
4
+
5
+ class MultiHeadAttention(Layer):
6
+
7
+ def __init__(
8
+ self,
9
+ d_model,
10
+ num_heads,
11
+ bias=True,
12
+ ):
13
+ super().__init__()
14
+
15
+ self.Wv = None
16
+ self.Wk = None
17
+ self.Wq = None
18
+ self.Wo = None
19
+ if d_model % num_heads != 0:
20
+ raise ValueError(
21
+ "d_model must be divisible by num_heads"
22
+ )
23
+
24
+ self.d_model = d_model
25
+ self.num_heads = num_heads
26
+ self.head_dim = d_model // num_heads
27
+
28
+ self.bias = bias
29
+
30
+ def build(self, input_shape):
31
+
32
+ self.Wq = self.add_weight(
33
+ shape=(self.d_model, self.d_model),
34
+ initializer="glorot_uniform",
35
+ name="Wq",
36
+ )
37
+
38
+ self.Wk = self.add_weight(
39
+ shape=(self.d_model, self.d_model),
40
+ initializer="glorot_uniform",
41
+ name="Wk",
42
+ )
43
+
44
+ self.Wv = self.add_weight(
45
+ shape=(self.d_model, self.d_model),
46
+ initializer="glorot_uniform",
47
+ name="Wv",
48
+ )
49
+
50
+ self.Wo = self.add_weight(
51
+ shape=(self.d_model, self.d_model),
52
+ initializer="glorot_uniform",
53
+ name="Wo",
54
+ )
55
+
56
+ def call(self, x, mask=None):
57
+
58
+ batch = x.shape[0]
59
+ seq = x.shape[1]
60
+
61
+ Q = x @ self.Wq
62
+ K = x @ self.Wk
63
+ V = x @ self.Wv
64
+
65
+ Q = Q.reshape(
66
+ batch,
67
+ seq,
68
+ self.num_heads,
69
+ self.head_dim,
70
+ )
71
+
72
+ K = K.reshape(
73
+ batch,
74
+ seq,
75
+ self.num_heads,
76
+ self.head_dim,
77
+ )
78
+
79
+ V = V.reshape(
80
+ batch,
81
+ seq,
82
+ self.num_heads,
83
+ self.head_dim,
84
+ )
85
+
86
+ Q = Q.transpose(0, 2, 1, 3)
87
+ K = K.transpose(0, 2, 1, 3)
88
+ V = V.transpose(0, 2, 1, 3)
89
+
90
+ scores = Q @ K.transpose(0, 1, 3, 2)
91
+
92
+ # scores = scores / math.sqrt(self.head_dim)
93
+ scores = scores / Tensor(self.head_dim).sqrt()
94
+ # mask = np.triu(
95
+ # np.ones((seq, seq)),
96
+ # k=1
97
+ # ).astype(bool)
98
+
99
+ if mask is not None:
100
+ scores = scores.masked_fill(mask == 0, -1e9)
101
+
102
+ weights = scores.softmax(axis=-1)
103
+
104
+ context = weights @ V
105
+
106
+ context = context.transpose(0, 2, 1, 3)
107
+
108
+ context = context.reshape(
109
+ batch,
110
+ seq,
111
+ self.d_model,
112
+ )
113
+
114
+ output = context @ self.Wo
115
+
116
+ return output
117
+
118
+ def state_dict(self):
119
+ return {
120
+ "Wq": self.Wq.data.copy(),
121
+ "Wk": self.Wk.data.copy(),
122
+ "Wv": self.Wv.data.copy(),
123
+ "Wo": self.Wo.data.copy(),
124
+ }
125
+
126
+ def load_state_dict(self, state):
127
+ self.Wq.data[:] = state["Wq"]
128
+ self.Wk.data[:] = state["Wk"]
129
+ self.Wv.data[:] = state["Wv"]
130
+ self.Wo.data[:] = state["Wo"]
@@ -0,0 +1,79 @@
1
+ from src.activations import activation_fns
2
+ from src.models.transformers.Dropout import Dropout
3
+ from src.models.transformers.LayerNorm import LayerNorm
4
+ from src.models.transformers.MultiHeadAttention import MultiHeadAttention
5
+ from src.neural.Dense import Dense
6
+ from src.neural.Layer import Layer
7
+
8
+
9
+ class TransformerBlock(Layer):
10
+
11
+ def __init__(self, d_model, num_heads, ff_dim):
12
+ super().__init__()
13
+ self.attn = MultiHeadAttention(d_model, num_heads)
14
+ self.norm1 = LayerNorm(d_model)
15
+
16
+ self.fc1 = Dense(ff_dim)
17
+ self.fc2 = Dense(d_model)
18
+
19
+ self.norm2 = LayerNorm(d_model)
20
+ self.dropout = Dropout(0.1)
21
+
22
+ def call(self, x):
23
+ h = self.attn(x)
24
+ x = self.norm1(x + self.dropout(h))
25
+ # x = self.norm1(x + h)
26
+
27
+ h = self.fc2(activation_fns["gelu"](self.fc1(x)))
28
+ x = self.norm2(x + self.dropout(h))
29
+ # x = self.norm2(x + h)
30
+
31
+ return x
32
+
33
+ def build(self, input_shape):
34
+ self.attn.build(input_shape)
35
+
36
+ self.norm1.build(input_shape)
37
+
38
+ self.fc1.build(input_shape)
39
+
40
+ ff_shape = (*input_shape[:-1], self.fc1.units)
41
+
42
+ self.fc2.build(ff_shape)
43
+
44
+ self.norm2.build(input_shape)
45
+
46
+ self.dropout.build(input_shape)
47
+
48
+ self.built = True
49
+
50
+ def parameters(self):
51
+ params = []
52
+
53
+ params.extend(self.attn.parameters())
54
+
55
+ params.extend(self.norm1.parameters())
56
+
57
+ params.extend(self.fc1.parameters())
58
+
59
+ params.extend(self.fc2.parameters())
60
+
61
+ params.extend(self.norm2.parameters())
62
+
63
+ return params
64
+
65
+ def state_dict(self):
66
+ return {
67
+ "attn": self.attn.state_dict(),
68
+ "norm1": self.norm1.state_dict(),
69
+ "fc1": self.fc1.state_dict(),
70
+ "fc2": self.fc2.state_dict(),
71
+ "norm2": self.norm2.state_dict(),
72
+ }
73
+
74
+ def load_state_dict(self, state):
75
+ self.attn.load_state_dict(state["attn"])
76
+ self.norm1.load_state_dict(state["norm1"])
77
+ self.fc1.load_state_dict(state["fc1"])
78
+ self.fc2.load_state_dict(state["fc2"])
79
+ self.norm2.load_state_dict(state["norm2"])
File without changes
src/neural/Dense.py ADDED
@@ -0,0 +1,58 @@
1
+ from src.neural.Layer import Layer
2
+ from src.activations import activation_fns
3
+
4
+
5
+ class Dense(Layer):
6
+ def __init__(self, units, activation=None,
7
+ kernel_initializer = "glorot_uniform",
8
+ bias_initializer = "zeros",
9
+ trainable = True,
10
+ ):
11
+ super().__init__()
12
+ self.units = units
13
+ self.activation = activation
14
+
15
+ self.kernel_initializer = kernel_initializer
16
+ self.bias_initializer = bias_initializer
17
+ self.trainable = trainable
18
+
19
+ self.kernel = None
20
+ self.bias = None
21
+
22
+ def build(self, input_shape):
23
+ in_features = input_shape[-1]
24
+
25
+ self.kernel = self.add_weight(
26
+ # row = in_features, column = units
27
+ shape=(in_features, self.units),
28
+ initializer=self.kernel_initializer,
29
+ trainable=self.trainable,
30
+ name="kernel",
31
+ )
32
+
33
+ self.bias = self.add_weight(
34
+ shape=(self.units,),
35
+ initializer=self.bias_initializer,
36
+ trainable=self.trainable,
37
+ name="bias",
38
+ )
39
+
40
+ def call(self, inputs):
41
+ output = inputs @ self.kernel
42
+ output = output + self.bias
43
+
44
+ if self.activation is not None:
45
+ output = activation_fns[self.activation](output)
46
+
47
+ return output
48
+
49
+ def compute_output_shape(self, input_shape):
50
+ return input_shape[:-1] + (self.units,)
51
+
52
+ def get_config(self):
53
+ return {
54
+ "units": self.units,
55
+ "activation": self.activation,
56
+ "kernel_initializer": self.kernel_initializer,
57
+ "bias_initializer": self.bias_initializer,
58
+ }
src/neural/LSTM.py ADDED
@@ -0,0 +1,167 @@
1
+ from src.core.Tensor import Tensor
2
+ from src.neural.Layer import Layer
3
+
4
+
5
+ class LSTM(Layer):
6
+
7
+ def __init__(
8
+ self,
9
+ hidden_size,
10
+ return_sequences=False,
11
+ ):
12
+ super().__init__()
13
+
14
+ self.bo = None
15
+ self.Uo = None
16
+ self.Wo = None
17
+ self.bg = None
18
+ self.Ug = None
19
+ self.Wg = None
20
+ self.bf = None
21
+ self.Uf = None
22
+ self.Wf = None
23
+ self.bi = None
24
+ self.Ui = None
25
+ self.Wi = None
26
+ self.hidden_size = hidden_size
27
+ self.return_sequences = return_sequences
28
+
29
+ def build(self, input_shape):
30
+
31
+ input_size = input_shape[-1]
32
+
33
+ self.Wi = self.add_weight(
34
+ shape=(input_size, self.hidden_size),
35
+ initializer="glorot_uniform",
36
+ name="Wi",
37
+ )
38
+
39
+ self.Ui = self.add_weight(
40
+ shape=(self.hidden_size, self.hidden_size),
41
+ initializer="orthogonal",
42
+ name="Ui",
43
+ )
44
+
45
+ self.bi = self.add_weight(
46
+ shape=(self.hidden_size,),
47
+ initializer="zeros",
48
+ name="bi",
49
+ )
50
+
51
+ self.Wf = self.add_weight(
52
+ shape=(input_size, self.hidden_size),
53
+ initializer="glorot_uniform",
54
+ name="Wf",
55
+ )
56
+
57
+ self.Uf = self.add_weight(
58
+ shape=(self.hidden_size, self.hidden_size),
59
+ initializer="orthogonal",
60
+ name="Uf",
61
+ )
62
+
63
+ self.bf = self.add_weight(
64
+ shape=(self.hidden_size,),
65
+ initializer="zeros",
66
+ name="bf",
67
+ )
68
+
69
+ self.Wg = self.add_weight(
70
+ shape=(input_size, self.hidden_size),
71
+ initializer="glorot_uniform",
72
+ name="Wg",
73
+ )
74
+
75
+ self.Ug = self.add_weight(
76
+ shape=(self.hidden_size, self.hidden_size),
77
+ initializer="orthogonal",
78
+ name="Ug",
79
+ )
80
+
81
+ self.bg = self.add_weight(
82
+ shape=(self.hidden_size,),
83
+ initializer="zeros",
84
+ name="bg",
85
+ )
86
+
87
+ self.Wo = self.add_weight(
88
+ shape=(input_size, self.hidden_size),
89
+ initializer="glorot_uniform",
90
+ name="Wo",
91
+ )
92
+
93
+ self.Uo = self.add_weight(
94
+ shape=(self.hidden_size, self.hidden_size),
95
+ initializer="orthogonal",
96
+ name="Uo",
97
+ )
98
+
99
+ self.bo = self.add_weight(
100
+ shape=(self.hidden_size,),
101
+ initializer="zeros",
102
+ name="bo",
103
+ )
104
+
105
+ def call(self, x):
106
+
107
+ batch = x.shape[0]
108
+ timesteps = x.shape[1]
109
+
110
+ h = Tensor.zeros(
111
+ (batch, self.hidden_size),
112
+ requires_grad=False,
113
+ )
114
+
115
+ c = Tensor.zeros(
116
+ (batch, self.hidden_size),
117
+ requires_grad=False,
118
+ )
119
+
120
+ outputs = []
121
+
122
+ for t in range(timesteps):
123
+
124
+ xt = x[:, t, :]
125
+
126
+ i = (
127
+ xt @ self.Wi +
128
+ h @ self.Ui +
129
+ self.bi
130
+ ).sigmoid()
131
+
132
+ f = (
133
+ xt @ self.Wf +
134
+ h @ self.Uf +
135
+ self.bf
136
+ ).sigmoid()
137
+
138
+ g = (
139
+ xt @ self.Wg +
140
+ h @ self.Ug +
141
+ self.bg
142
+ ).tanh()
143
+
144
+ o = (
145
+ xt @ self.Wo +
146
+ h @ self.Uo +
147
+ self.bo
148
+ ).sigmoid()
149
+
150
+ c = f * c + i * g
151
+
152
+ h = o * c.tanh()
153
+
154
+ if self.return_sequences:
155
+ outputs.append(h)
156
+
157
+ if self.return_sequences:
158
+ return Tensor.stack(outputs, axis=1)
159
+
160
+ return h
161
+
162
+ def get_config(self):
163
+ return {
164
+ "class_name": "LSTM",
165
+ "hidden_size": self.hidden_size,
166
+ "return_sequences": self.return_sequences,
167
+ }
src/neural/Layer.py ADDED
@@ -0,0 +1,72 @@
1
+ from src.neural.Parameter import Parameter
2
+ from src.initializers import initializer_fns
3
+
4
+ class Layer:
5
+ def __init__(self):
6
+ self.built = False
7
+
8
+ self.weights = []
9
+ self.trainable_weights = []
10
+ self.non_trainable_weights = []
11
+
12
+ def __call__(self, inputs):
13
+ if not self.built:
14
+ self.build(inputs.shape)
15
+ self.built = True
16
+
17
+ return self.call(inputs)
18
+
19
+ def build(self, input_shape):
20
+ raise NotImplementedError
21
+
22
+ def call(self, inputs):
23
+ raise NotImplementedError
24
+
25
+ def add_weight(
26
+ self,
27
+ shape,
28
+ initializer="glorot_uniform",
29
+ trainable=True,
30
+ name=None,
31
+ ):
32
+
33
+ value = initializer_fns[initializer](shape)
34
+
35
+ parameter = Parameter(
36
+ data=value,
37
+ trainable=trainable,
38
+ name=name,
39
+ )
40
+
41
+ self.weights.append(parameter)
42
+
43
+ if trainable:
44
+ self.trainable_weights.append(parameter)
45
+ else:
46
+ self.non_trainable_weights.append(parameter)
47
+
48
+ return parameter
49
+
50
+ def parameters(self):
51
+ return self.trainable_weights
52
+
53
+ def state_dict(self):
54
+ state = {}
55
+
56
+ for p in self.parameters():
57
+ state[p.name] = p.data.copy()
58
+
59
+ return state
60
+
61
+ def load_state_dict(self, state):
62
+
63
+ if not state:
64
+ return
65
+
66
+ for p in self.parameters():
67
+
68
+ if p.name in state:
69
+ p.data[:] = state[p.name]
70
+
71
+ def compute_output_shape(self, input_shape):
72
+ raise NotImplementedError
@@ -0,0 +1,30 @@
1
+ import numpy as np
2
+
3
+ from src.core.Tensor import Tensor
4
+
5
+
6
+ # class Parameter:
7
+ # def __init__(self, data, trainable=True, name=None):
8
+ # self.data = np.asarray(data, dtype=np.float32)
9
+ # self.grad = np.zeros_like(self.data)
10
+ #
11
+ # self.trainable = trainable
12
+ # self.name = name
13
+ #
14
+ # def zero_grad(self):
15
+ # self.grad.fill(0)
16
+ #
17
+ # @property
18
+ # def shape(self):
19
+ # return self.data.shape
20
+
21
+ class Parameter(Tensor):
22
+
23
+ def __init__(self, data, trainable=True, name=None):
24
+ super().__init__(
25
+ data,
26
+ requires_grad=True,
27
+ )
28
+
29
+ self.name = name
30
+ self.trainable = trainable
src/neural/RNN.py ADDED
@@ -0,0 +1,83 @@
1
+ from src.neural.Layer import Layer
2
+ from src.activations import Tanh
3
+ from src.core.Tensor import Tensor
4
+
5
+
6
+ class RNN(Layer):
7
+
8
+ def __init__(
9
+ self,
10
+ hidden_size,
11
+ return_sequences=False,
12
+ ):
13
+ super().__init__()
14
+
15
+ self.hidden_size = hidden_size
16
+ self.return_sequences = return_sequences
17
+
18
+ self.Wx = None
19
+ self.Wh = None
20
+ self.bh = None
21
+
22
+ def build(self, input_shape):
23
+ input_size = input_shape[-1]
24
+
25
+ self.Wx = self.add_weight(
26
+ shape=(input_size, self.hidden_size),
27
+ initializer="glorot_uniform",
28
+ trainable=True,
29
+ name="Wx",
30
+ )
31
+
32
+ self.Wh = self.add_weight(
33
+ shape=(self.hidden_size, self.hidden_size),
34
+ initializer="orthogonal",
35
+ trainable=True,
36
+ name="Wh",
37
+ )
38
+
39
+ self.bh = self.add_weight(
40
+ shape=(self.hidden_size,),
41
+ initializer="zeros",
42
+ trainable=True,
43
+ name="bh",
44
+ )
45
+
46
+ def call(self, x):
47
+
48
+ batch_size = x.shape[0]
49
+ time_steps = x.shape[1]
50
+
51
+ h = Tensor.zeros(
52
+ (batch_size, self.hidden_size),
53
+ requires_grad=False,
54
+ )
55
+
56
+ outputs = []
57
+
58
+ for t in range(time_steps):
59
+
60
+ xt = x[:, t, :]
61
+
62
+ h = Tanh.forward(
63
+ xt @ self.Wx +
64
+ h @ self.Wh +
65
+ self.bh
66
+ )
67
+
68
+ if self.return_sequences:
69
+ outputs.append(h)
70
+
71
+ if self.return_sequences:
72
+ return Tensor.stack(outputs, axis=1)
73
+
74
+ return h
75
+
76
+ def compute_output_shape(self, input_shape):
77
+
78
+ batch = input_shape[0]
79
+
80
+ return (
81
+ batch,
82
+ self.hidden_size,
83
+ )
src/neural/__init__.py ADDED
File without changes
src/ops/__init__.py ADDED
File without changes
src/ops/stack.py ADDED
@@ -0,0 +1,40 @@
1
+ import numpy as np
2
+
3
+ from src.core.Tensor import Tensor
4
+
5
+
6
+ class Stack:
7
+
8
+ @staticmethod
9
+ def forward(tensors, axis=0):
10
+
11
+ data = np.stack(
12
+ [t.data for t in tensors],
13
+ axis=axis,
14
+ )
15
+
16
+ out = Tensor(
17
+ data,
18
+ requires_grad=any(t.requires_grad for t in tensors),
19
+ parents=tuple(tensors),
20
+ op="Stack",
21
+ )
22
+
23
+ def _backward():
24
+
25
+ for i, tensor in enumerate(tensors):
26
+
27
+ if not tensor.requires_grad:
28
+ continue
29
+
30
+ grad = np.take(
31
+ out.grad,
32
+ indices=i,
33
+ axis=axis,
34
+ )
35
+
36
+ tensor.grad += grad
37
+
38
+ out._backward = _backward
39
+
40
+ return out
@@ -0,0 +1,31 @@
1
+ import numpy as np
2
+
3
+
4
+ class Adagrad:
5
+
6
+ def __init__(self, lr=0.01, eps=1e-8):
7
+ self.lr = lr
8
+ self.eps = eps
9
+ self.cache = {}
10
+
11
+ def step(self, params):
12
+
13
+ for p in params:
14
+
15
+ if not p.requires_grad:
16
+ continue
17
+
18
+ if id(p) not in self.cache:
19
+ self.cache[id(p)] = np.zeros_like(p.data)
20
+
21
+ self.cache[id(p)] += p.grad ** 2
22
+
23
+ p.data -= (
24
+ self.lr
25
+ * p.grad
26
+ / (np.sqrt(self.cache[id(p)]) + self.eps)
27
+ )
28
+
29
+ def zero_grad(self, params):
30
+ for p in params:
31
+ p.zero_grad()