pytensorforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli.py +604 -0
- pytensorforge-0.1.0.dist-info/METADATA +103 -0
- pytensorforge-0.1.0.dist-info/RECORD +146 -0
- pytensorforge-0.1.0.dist-info/WHEEL +5 -0
- pytensorforge-0.1.0.dist-info/entry_points.txt +2 -0
- pytensorforge-0.1.0.dist-info/top_level.txt +2 -0
- src/__init__.py +0 -0
- src/activations/Activation.py +4 -0
- src/activations/ELU.py +11 -0
- src/activations/GELU.py +6 -0
- src/activations/ReLU.py +27 -0
- src/activations/SELU.py +14 -0
- src/activations/Sigmoid.py +27 -0
- src/activations/Softmax.py +84 -0
- src/activations/Tanh.py +29 -0
- src/activations/__init__.py +17 -0
- src/config.py +120 -0
- src/core/Matrix.py +3 -0
- src/core/Scalar.py +18 -0
- src/core/Tensor.py +866 -0
- src/core/Vector.py +31 -0
- src/core/__init__.py +0 -0
- src/data/__init__.py +0 -0
- src/data/chat_dataset.py +188 -0
- src/data/corpus.py +104 -0
- src/data/document_stream.py +178 -0
- src/data/parallel_encode.py +86 -0
- src/data/prefetch.py +62 -0
- src/data/shard_builder.py +119 -0
- src/data/shard_writer.py +81 -0
- src/data/sharded_dataset.py +112 -0
- src/data/streaming_dataset.py +132 -0
- src/data/validation.py +212 -0
- src/inference/__init__.py +0 -0
- src/inference/chat_template.py +384 -0
- src/inference/config.py +48 -0
- src/inference/engine.py +241 -0
- src/inference/export.py +133 -0
- src/inference/kv_cache.py +65 -0
- src/inference/runtime.py +161 -0
- src/inference/sampling.py +42 -0
- src/inference/scheduler.py +473 -0
- src/inference/text.py +67 -0
- src/initializers/Constant.py +9 -0
- src/initializers/GlorotNormal.py +15 -0
- src/initializers/GlorotUniform.py +26 -0
- src/initializers/HeNormal.py +15 -0
- src/initializers/HeUniform.py +14 -0
- src/initializers/Initializer.py +4 -0
- src/initializers/LecunNormal.py +16 -0
- src/initializers/LecunUniform.py +14 -0
- src/initializers/Ones.py +6 -0
- src/initializers/Orthogonal.py +14 -0
- src/initializers/RandomNormal.py +14 -0
- src/initializers/RandomUniform.py +14 -0
- src/initializers/Zeros.py +8 -0
- src/initializers/__init__.py +17 -0
- src/loss/CategoricalCrossEntropy.py +9 -0
- src/loss/CrossEntropyLoss.py +34 -0
- src/loss/CrossEntropyWithLogitsLoss.py +59 -0
- src/loss/Hinge.py +5 -0
- src/loss/Huber.py +22 -0
- src/loss/Loss.py +6 -0
- src/loss/MSE.py +7 -0
- src/loss/MSELoss.py +10 -0
- src/loss/SparseCategoricalCrossEntropy.py +15 -0
- src/loss/__init__.py +18 -0
- src/loss/bce.py +34 -0
- src/loss/mae.py +16 -0
- src/math/__init__.py +0 -0
- src/math/clip.py +37 -0
- src/math/exp.py +27 -0
- src/math/log.py +25 -0
- src/math/sigmoid.py +5 -0
- src/models/__init__.py +0 -0
- src/models/embedding/Embedding.py +65 -0
- src/models/embedding/__init__.py +0 -0
- src/models/gpt/__init__.py +0 -0
- src/models/gpt/attention.py +158 -0
- src/models/gpt/block.py +74 -0
- src/models/gpt/config.py +103 -0
- src/models/gpt/context.py +44 -0
- src/models/gpt/model.py +165 -0
- src/models/gpt/recompute.py +35 -0
- src/models/gpt/rope.py +84 -0
- src/models/regression/Linear.py +51 -0
- src/models/regression/Logistic.py +36 -0
- src/models/regression/__init__.py +0 -0
- src/models/seq/Sequential.py +297 -0
- src/models/seq/__init__.py +0 -0
- src/models/svm/__init__.py +0 -0
- src/models/tokenizer/BPETokenizer.py +228 -0
- src/models/tokenizer/__init__.py +0 -0
- src/models/transformers/Dropout.py +35 -0
- src/models/transformers/LastToken.py +10 -0
- src/models/transformers/LayerNorm.py +54 -0
- src/models/transformers/Linear.py +18 -0
- src/models/transformers/MultiHeadAttention.py +130 -0
- src/models/transformers/TransformerBlock.py +79 -0
- src/models/transformers/__init__.py +0 -0
- src/neural/Dense.py +58 -0
- src/neural/LSTM.py +167 -0
- src/neural/Layer.py +72 -0
- src/neural/Parameter.py +30 -0
- src/neural/RNN.py +83 -0
- src/neural/__init__.py +0 -0
- src/ops/__init__.py +0 -0
- src/ops/stack.py +40 -0
- src/optimizers/Adagrad.py +31 -0
- src/optimizers/Adam.py +98 -0
- src/optimizers/AdamW.py +84 -0
- src/optimizers/Batch.py +11 -0
- src/optimizers/Nesterov.py +35 -0
- src/optimizers/Optimizer.py +18 -0
- src/optimizers/RMSProp.py +35 -0
- src/optimizers/SGD.py +30 -0
- src/optimizers/SGDMomentum.py +28 -0
- src/optimizers/__init__.py +9 -0
- src/scaling/StandardScaler.py +15 -0
- src/scaling/__init__.py +0 -0
- src/serialization/__init__.py +0 -0
- src/serialization/checkpoint.py +58 -0
- src/serialization/modelio.py +132 -0
- src/serving/__init__.py +0 -0
- src/serving/app.py +792 -0
- src/serving/config.py +216 -0
- src/serving/errors.py +51 -0
- src/serving/http.py +599 -0
- src/serving/metrics.py +293 -0
- src/serving/model_server.py +287 -0
- src/serving/protocol.py +377 -0
- src/serving/security.py +200 -0
- src/serving/server.py +121 -0
- src/tokenization/__init__.py +0 -0
- src/tokenization/base.py +75 -0
- src/tokenization/bpe.py +190 -0
- src/tokenization/bytebpe.py +476 -0
- src/tokenization/registry.py +28 -0
- src/training/__init__.py +0 -0
- src/training/checkpoint_manager.py +101 -0
- src/training/experiment.py +71 -0
- src/training/losses.py +42 -0
- src/training/precision.py +141 -0
- src/training/profiler.py +38 -0
- src/training/scheduler.py +50 -0
- src/training/trainer.py +594 -0
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
from src.core.Tensor import Tensor
|
|
2
|
+
from src.neural.Layer import Layer
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class MultiHeadAttention(Layer):
|
|
6
|
+
|
|
7
|
+
def __init__(
|
|
8
|
+
self,
|
|
9
|
+
d_model,
|
|
10
|
+
num_heads,
|
|
11
|
+
bias=True,
|
|
12
|
+
):
|
|
13
|
+
super().__init__()
|
|
14
|
+
|
|
15
|
+
self.Wv = None
|
|
16
|
+
self.Wk = None
|
|
17
|
+
self.Wq = None
|
|
18
|
+
self.Wo = None
|
|
19
|
+
if d_model % num_heads != 0:
|
|
20
|
+
raise ValueError(
|
|
21
|
+
"d_model must be divisible by num_heads"
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
self.d_model = d_model
|
|
25
|
+
self.num_heads = num_heads
|
|
26
|
+
self.head_dim = d_model // num_heads
|
|
27
|
+
|
|
28
|
+
self.bias = bias
|
|
29
|
+
|
|
30
|
+
def build(self, input_shape):
|
|
31
|
+
|
|
32
|
+
self.Wq = self.add_weight(
|
|
33
|
+
shape=(self.d_model, self.d_model),
|
|
34
|
+
initializer="glorot_uniform",
|
|
35
|
+
name="Wq",
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
self.Wk = self.add_weight(
|
|
39
|
+
shape=(self.d_model, self.d_model),
|
|
40
|
+
initializer="glorot_uniform",
|
|
41
|
+
name="Wk",
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
self.Wv = self.add_weight(
|
|
45
|
+
shape=(self.d_model, self.d_model),
|
|
46
|
+
initializer="glorot_uniform",
|
|
47
|
+
name="Wv",
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
self.Wo = self.add_weight(
|
|
51
|
+
shape=(self.d_model, self.d_model),
|
|
52
|
+
initializer="glorot_uniform",
|
|
53
|
+
name="Wo",
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
def call(self, x, mask=None):
|
|
57
|
+
|
|
58
|
+
batch = x.shape[0]
|
|
59
|
+
seq = x.shape[1]
|
|
60
|
+
|
|
61
|
+
Q = x @ self.Wq
|
|
62
|
+
K = x @ self.Wk
|
|
63
|
+
V = x @ self.Wv
|
|
64
|
+
|
|
65
|
+
Q = Q.reshape(
|
|
66
|
+
batch,
|
|
67
|
+
seq,
|
|
68
|
+
self.num_heads,
|
|
69
|
+
self.head_dim,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
K = K.reshape(
|
|
73
|
+
batch,
|
|
74
|
+
seq,
|
|
75
|
+
self.num_heads,
|
|
76
|
+
self.head_dim,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
V = V.reshape(
|
|
80
|
+
batch,
|
|
81
|
+
seq,
|
|
82
|
+
self.num_heads,
|
|
83
|
+
self.head_dim,
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
Q = Q.transpose(0, 2, 1, 3)
|
|
87
|
+
K = K.transpose(0, 2, 1, 3)
|
|
88
|
+
V = V.transpose(0, 2, 1, 3)
|
|
89
|
+
|
|
90
|
+
scores = Q @ K.transpose(0, 1, 3, 2)
|
|
91
|
+
|
|
92
|
+
# scores = scores / math.sqrt(self.head_dim)
|
|
93
|
+
scores = scores / Tensor(self.head_dim).sqrt()
|
|
94
|
+
# mask = np.triu(
|
|
95
|
+
# np.ones((seq, seq)),
|
|
96
|
+
# k=1
|
|
97
|
+
# ).astype(bool)
|
|
98
|
+
|
|
99
|
+
if mask is not None:
|
|
100
|
+
scores = scores.masked_fill(mask == 0, -1e9)
|
|
101
|
+
|
|
102
|
+
weights = scores.softmax(axis=-1)
|
|
103
|
+
|
|
104
|
+
context = weights @ V
|
|
105
|
+
|
|
106
|
+
context = context.transpose(0, 2, 1, 3)
|
|
107
|
+
|
|
108
|
+
context = context.reshape(
|
|
109
|
+
batch,
|
|
110
|
+
seq,
|
|
111
|
+
self.d_model,
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
output = context @ self.Wo
|
|
115
|
+
|
|
116
|
+
return output
|
|
117
|
+
|
|
118
|
+
def state_dict(self):
|
|
119
|
+
return {
|
|
120
|
+
"Wq": self.Wq.data.copy(),
|
|
121
|
+
"Wk": self.Wk.data.copy(),
|
|
122
|
+
"Wv": self.Wv.data.copy(),
|
|
123
|
+
"Wo": self.Wo.data.copy(),
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
def load_state_dict(self, state):
|
|
127
|
+
self.Wq.data[:] = state["Wq"]
|
|
128
|
+
self.Wk.data[:] = state["Wk"]
|
|
129
|
+
self.Wv.data[:] = state["Wv"]
|
|
130
|
+
self.Wo.data[:] = state["Wo"]
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
from src.activations import activation_fns
|
|
2
|
+
from src.models.transformers.Dropout import Dropout
|
|
3
|
+
from src.models.transformers.LayerNorm import LayerNorm
|
|
4
|
+
from src.models.transformers.MultiHeadAttention import MultiHeadAttention
|
|
5
|
+
from src.neural.Dense import Dense
|
|
6
|
+
from src.neural.Layer import Layer
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class TransformerBlock(Layer):
|
|
10
|
+
|
|
11
|
+
def __init__(self, d_model, num_heads, ff_dim):
|
|
12
|
+
super().__init__()
|
|
13
|
+
self.attn = MultiHeadAttention(d_model, num_heads)
|
|
14
|
+
self.norm1 = LayerNorm(d_model)
|
|
15
|
+
|
|
16
|
+
self.fc1 = Dense(ff_dim)
|
|
17
|
+
self.fc2 = Dense(d_model)
|
|
18
|
+
|
|
19
|
+
self.norm2 = LayerNorm(d_model)
|
|
20
|
+
self.dropout = Dropout(0.1)
|
|
21
|
+
|
|
22
|
+
def call(self, x):
|
|
23
|
+
h = self.attn(x)
|
|
24
|
+
x = self.norm1(x + self.dropout(h))
|
|
25
|
+
# x = self.norm1(x + h)
|
|
26
|
+
|
|
27
|
+
h = self.fc2(activation_fns["gelu"](self.fc1(x)))
|
|
28
|
+
x = self.norm2(x + self.dropout(h))
|
|
29
|
+
# x = self.norm2(x + h)
|
|
30
|
+
|
|
31
|
+
return x
|
|
32
|
+
|
|
33
|
+
def build(self, input_shape):
|
|
34
|
+
self.attn.build(input_shape)
|
|
35
|
+
|
|
36
|
+
self.norm1.build(input_shape)
|
|
37
|
+
|
|
38
|
+
self.fc1.build(input_shape)
|
|
39
|
+
|
|
40
|
+
ff_shape = (*input_shape[:-1], self.fc1.units)
|
|
41
|
+
|
|
42
|
+
self.fc2.build(ff_shape)
|
|
43
|
+
|
|
44
|
+
self.norm2.build(input_shape)
|
|
45
|
+
|
|
46
|
+
self.dropout.build(input_shape)
|
|
47
|
+
|
|
48
|
+
self.built = True
|
|
49
|
+
|
|
50
|
+
def parameters(self):
|
|
51
|
+
params = []
|
|
52
|
+
|
|
53
|
+
params.extend(self.attn.parameters())
|
|
54
|
+
|
|
55
|
+
params.extend(self.norm1.parameters())
|
|
56
|
+
|
|
57
|
+
params.extend(self.fc1.parameters())
|
|
58
|
+
|
|
59
|
+
params.extend(self.fc2.parameters())
|
|
60
|
+
|
|
61
|
+
params.extend(self.norm2.parameters())
|
|
62
|
+
|
|
63
|
+
return params
|
|
64
|
+
|
|
65
|
+
def state_dict(self):
|
|
66
|
+
return {
|
|
67
|
+
"attn": self.attn.state_dict(),
|
|
68
|
+
"norm1": self.norm1.state_dict(),
|
|
69
|
+
"fc1": self.fc1.state_dict(),
|
|
70
|
+
"fc2": self.fc2.state_dict(),
|
|
71
|
+
"norm2": self.norm2.state_dict(),
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
def load_state_dict(self, state):
|
|
75
|
+
self.attn.load_state_dict(state["attn"])
|
|
76
|
+
self.norm1.load_state_dict(state["norm1"])
|
|
77
|
+
self.fc1.load_state_dict(state["fc1"])
|
|
78
|
+
self.fc2.load_state_dict(state["fc2"])
|
|
79
|
+
self.norm2.load_state_dict(state["norm2"])
|
|
File without changes
|
src/neural/Dense.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
from src.neural.Layer import Layer
|
|
2
|
+
from src.activations import activation_fns
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class Dense(Layer):
|
|
6
|
+
def __init__(self, units, activation=None,
|
|
7
|
+
kernel_initializer = "glorot_uniform",
|
|
8
|
+
bias_initializer = "zeros",
|
|
9
|
+
trainable = True,
|
|
10
|
+
):
|
|
11
|
+
super().__init__()
|
|
12
|
+
self.units = units
|
|
13
|
+
self.activation = activation
|
|
14
|
+
|
|
15
|
+
self.kernel_initializer = kernel_initializer
|
|
16
|
+
self.bias_initializer = bias_initializer
|
|
17
|
+
self.trainable = trainable
|
|
18
|
+
|
|
19
|
+
self.kernel = None
|
|
20
|
+
self.bias = None
|
|
21
|
+
|
|
22
|
+
def build(self, input_shape):
|
|
23
|
+
in_features = input_shape[-1]
|
|
24
|
+
|
|
25
|
+
self.kernel = self.add_weight(
|
|
26
|
+
# row = in_features, column = units
|
|
27
|
+
shape=(in_features, self.units),
|
|
28
|
+
initializer=self.kernel_initializer,
|
|
29
|
+
trainable=self.trainable,
|
|
30
|
+
name="kernel",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
self.bias = self.add_weight(
|
|
34
|
+
shape=(self.units,),
|
|
35
|
+
initializer=self.bias_initializer,
|
|
36
|
+
trainable=self.trainable,
|
|
37
|
+
name="bias",
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
def call(self, inputs):
|
|
41
|
+
output = inputs @ self.kernel
|
|
42
|
+
output = output + self.bias
|
|
43
|
+
|
|
44
|
+
if self.activation is not None:
|
|
45
|
+
output = activation_fns[self.activation](output)
|
|
46
|
+
|
|
47
|
+
return output
|
|
48
|
+
|
|
49
|
+
def compute_output_shape(self, input_shape):
|
|
50
|
+
return input_shape[:-1] + (self.units,)
|
|
51
|
+
|
|
52
|
+
def get_config(self):
|
|
53
|
+
return {
|
|
54
|
+
"units": self.units,
|
|
55
|
+
"activation": self.activation,
|
|
56
|
+
"kernel_initializer": self.kernel_initializer,
|
|
57
|
+
"bias_initializer": self.bias_initializer,
|
|
58
|
+
}
|
src/neural/LSTM.py
ADDED
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
from src.core.Tensor import Tensor
|
|
2
|
+
from src.neural.Layer import Layer
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class LSTM(Layer):
|
|
6
|
+
|
|
7
|
+
def __init__(
|
|
8
|
+
self,
|
|
9
|
+
hidden_size,
|
|
10
|
+
return_sequences=False,
|
|
11
|
+
):
|
|
12
|
+
super().__init__()
|
|
13
|
+
|
|
14
|
+
self.bo = None
|
|
15
|
+
self.Uo = None
|
|
16
|
+
self.Wo = None
|
|
17
|
+
self.bg = None
|
|
18
|
+
self.Ug = None
|
|
19
|
+
self.Wg = None
|
|
20
|
+
self.bf = None
|
|
21
|
+
self.Uf = None
|
|
22
|
+
self.Wf = None
|
|
23
|
+
self.bi = None
|
|
24
|
+
self.Ui = None
|
|
25
|
+
self.Wi = None
|
|
26
|
+
self.hidden_size = hidden_size
|
|
27
|
+
self.return_sequences = return_sequences
|
|
28
|
+
|
|
29
|
+
def build(self, input_shape):
|
|
30
|
+
|
|
31
|
+
input_size = input_shape[-1]
|
|
32
|
+
|
|
33
|
+
self.Wi = self.add_weight(
|
|
34
|
+
shape=(input_size, self.hidden_size),
|
|
35
|
+
initializer="glorot_uniform",
|
|
36
|
+
name="Wi",
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
self.Ui = self.add_weight(
|
|
40
|
+
shape=(self.hidden_size, self.hidden_size),
|
|
41
|
+
initializer="orthogonal",
|
|
42
|
+
name="Ui",
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
self.bi = self.add_weight(
|
|
46
|
+
shape=(self.hidden_size,),
|
|
47
|
+
initializer="zeros",
|
|
48
|
+
name="bi",
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
self.Wf = self.add_weight(
|
|
52
|
+
shape=(input_size, self.hidden_size),
|
|
53
|
+
initializer="glorot_uniform",
|
|
54
|
+
name="Wf",
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
self.Uf = self.add_weight(
|
|
58
|
+
shape=(self.hidden_size, self.hidden_size),
|
|
59
|
+
initializer="orthogonal",
|
|
60
|
+
name="Uf",
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
self.bf = self.add_weight(
|
|
64
|
+
shape=(self.hidden_size,),
|
|
65
|
+
initializer="zeros",
|
|
66
|
+
name="bf",
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
self.Wg = self.add_weight(
|
|
70
|
+
shape=(input_size, self.hidden_size),
|
|
71
|
+
initializer="glorot_uniform",
|
|
72
|
+
name="Wg",
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
self.Ug = self.add_weight(
|
|
76
|
+
shape=(self.hidden_size, self.hidden_size),
|
|
77
|
+
initializer="orthogonal",
|
|
78
|
+
name="Ug",
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
self.bg = self.add_weight(
|
|
82
|
+
shape=(self.hidden_size,),
|
|
83
|
+
initializer="zeros",
|
|
84
|
+
name="bg",
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
self.Wo = self.add_weight(
|
|
88
|
+
shape=(input_size, self.hidden_size),
|
|
89
|
+
initializer="glorot_uniform",
|
|
90
|
+
name="Wo",
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
self.Uo = self.add_weight(
|
|
94
|
+
shape=(self.hidden_size, self.hidden_size),
|
|
95
|
+
initializer="orthogonal",
|
|
96
|
+
name="Uo",
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
self.bo = self.add_weight(
|
|
100
|
+
shape=(self.hidden_size,),
|
|
101
|
+
initializer="zeros",
|
|
102
|
+
name="bo",
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
def call(self, x):
|
|
106
|
+
|
|
107
|
+
batch = x.shape[0]
|
|
108
|
+
timesteps = x.shape[1]
|
|
109
|
+
|
|
110
|
+
h = Tensor.zeros(
|
|
111
|
+
(batch, self.hidden_size),
|
|
112
|
+
requires_grad=False,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
c = Tensor.zeros(
|
|
116
|
+
(batch, self.hidden_size),
|
|
117
|
+
requires_grad=False,
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
outputs = []
|
|
121
|
+
|
|
122
|
+
for t in range(timesteps):
|
|
123
|
+
|
|
124
|
+
xt = x[:, t, :]
|
|
125
|
+
|
|
126
|
+
i = (
|
|
127
|
+
xt @ self.Wi +
|
|
128
|
+
h @ self.Ui +
|
|
129
|
+
self.bi
|
|
130
|
+
).sigmoid()
|
|
131
|
+
|
|
132
|
+
f = (
|
|
133
|
+
xt @ self.Wf +
|
|
134
|
+
h @ self.Uf +
|
|
135
|
+
self.bf
|
|
136
|
+
).sigmoid()
|
|
137
|
+
|
|
138
|
+
g = (
|
|
139
|
+
xt @ self.Wg +
|
|
140
|
+
h @ self.Ug +
|
|
141
|
+
self.bg
|
|
142
|
+
).tanh()
|
|
143
|
+
|
|
144
|
+
o = (
|
|
145
|
+
xt @ self.Wo +
|
|
146
|
+
h @ self.Uo +
|
|
147
|
+
self.bo
|
|
148
|
+
).sigmoid()
|
|
149
|
+
|
|
150
|
+
c = f * c + i * g
|
|
151
|
+
|
|
152
|
+
h = o * c.tanh()
|
|
153
|
+
|
|
154
|
+
if self.return_sequences:
|
|
155
|
+
outputs.append(h)
|
|
156
|
+
|
|
157
|
+
if self.return_sequences:
|
|
158
|
+
return Tensor.stack(outputs, axis=1)
|
|
159
|
+
|
|
160
|
+
return h
|
|
161
|
+
|
|
162
|
+
def get_config(self):
|
|
163
|
+
return {
|
|
164
|
+
"class_name": "LSTM",
|
|
165
|
+
"hidden_size": self.hidden_size,
|
|
166
|
+
"return_sequences": self.return_sequences,
|
|
167
|
+
}
|
src/neural/Layer.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
from src.neural.Parameter import Parameter
|
|
2
|
+
from src.initializers import initializer_fns
|
|
3
|
+
|
|
4
|
+
class Layer:
|
|
5
|
+
def __init__(self):
|
|
6
|
+
self.built = False
|
|
7
|
+
|
|
8
|
+
self.weights = []
|
|
9
|
+
self.trainable_weights = []
|
|
10
|
+
self.non_trainable_weights = []
|
|
11
|
+
|
|
12
|
+
def __call__(self, inputs):
|
|
13
|
+
if not self.built:
|
|
14
|
+
self.build(inputs.shape)
|
|
15
|
+
self.built = True
|
|
16
|
+
|
|
17
|
+
return self.call(inputs)
|
|
18
|
+
|
|
19
|
+
def build(self, input_shape):
|
|
20
|
+
raise NotImplementedError
|
|
21
|
+
|
|
22
|
+
def call(self, inputs):
|
|
23
|
+
raise NotImplementedError
|
|
24
|
+
|
|
25
|
+
def add_weight(
|
|
26
|
+
self,
|
|
27
|
+
shape,
|
|
28
|
+
initializer="glorot_uniform",
|
|
29
|
+
trainable=True,
|
|
30
|
+
name=None,
|
|
31
|
+
):
|
|
32
|
+
|
|
33
|
+
value = initializer_fns[initializer](shape)
|
|
34
|
+
|
|
35
|
+
parameter = Parameter(
|
|
36
|
+
data=value,
|
|
37
|
+
trainable=trainable,
|
|
38
|
+
name=name,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
self.weights.append(parameter)
|
|
42
|
+
|
|
43
|
+
if trainable:
|
|
44
|
+
self.trainable_weights.append(parameter)
|
|
45
|
+
else:
|
|
46
|
+
self.non_trainable_weights.append(parameter)
|
|
47
|
+
|
|
48
|
+
return parameter
|
|
49
|
+
|
|
50
|
+
def parameters(self):
|
|
51
|
+
return self.trainable_weights
|
|
52
|
+
|
|
53
|
+
def state_dict(self):
|
|
54
|
+
state = {}
|
|
55
|
+
|
|
56
|
+
for p in self.parameters():
|
|
57
|
+
state[p.name] = p.data.copy()
|
|
58
|
+
|
|
59
|
+
return state
|
|
60
|
+
|
|
61
|
+
def load_state_dict(self, state):
|
|
62
|
+
|
|
63
|
+
if not state:
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
for p in self.parameters():
|
|
67
|
+
|
|
68
|
+
if p.name in state:
|
|
69
|
+
p.data[:] = state[p.name]
|
|
70
|
+
|
|
71
|
+
def compute_output_shape(self, input_shape):
|
|
72
|
+
raise NotImplementedError
|
src/neural/Parameter.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
|
|
3
|
+
from src.core.Tensor import Tensor
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
# class Parameter:
|
|
7
|
+
# def __init__(self, data, trainable=True, name=None):
|
|
8
|
+
# self.data = np.asarray(data, dtype=np.float32)
|
|
9
|
+
# self.grad = np.zeros_like(self.data)
|
|
10
|
+
#
|
|
11
|
+
# self.trainable = trainable
|
|
12
|
+
# self.name = name
|
|
13
|
+
#
|
|
14
|
+
# def zero_grad(self):
|
|
15
|
+
# self.grad.fill(0)
|
|
16
|
+
#
|
|
17
|
+
# @property
|
|
18
|
+
# def shape(self):
|
|
19
|
+
# return self.data.shape
|
|
20
|
+
|
|
21
|
+
class Parameter(Tensor):
|
|
22
|
+
|
|
23
|
+
def __init__(self, data, trainable=True, name=None):
|
|
24
|
+
super().__init__(
|
|
25
|
+
data,
|
|
26
|
+
requires_grad=True,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
self.name = name
|
|
30
|
+
self.trainable = trainable
|
src/neural/RNN.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
from src.neural.Layer import Layer
|
|
2
|
+
from src.activations import Tanh
|
|
3
|
+
from src.core.Tensor import Tensor
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class RNN(Layer):
|
|
7
|
+
|
|
8
|
+
def __init__(
|
|
9
|
+
self,
|
|
10
|
+
hidden_size,
|
|
11
|
+
return_sequences=False,
|
|
12
|
+
):
|
|
13
|
+
super().__init__()
|
|
14
|
+
|
|
15
|
+
self.hidden_size = hidden_size
|
|
16
|
+
self.return_sequences = return_sequences
|
|
17
|
+
|
|
18
|
+
self.Wx = None
|
|
19
|
+
self.Wh = None
|
|
20
|
+
self.bh = None
|
|
21
|
+
|
|
22
|
+
def build(self, input_shape):
|
|
23
|
+
input_size = input_shape[-1]
|
|
24
|
+
|
|
25
|
+
self.Wx = self.add_weight(
|
|
26
|
+
shape=(input_size, self.hidden_size),
|
|
27
|
+
initializer="glorot_uniform",
|
|
28
|
+
trainable=True,
|
|
29
|
+
name="Wx",
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
self.Wh = self.add_weight(
|
|
33
|
+
shape=(self.hidden_size, self.hidden_size),
|
|
34
|
+
initializer="orthogonal",
|
|
35
|
+
trainable=True,
|
|
36
|
+
name="Wh",
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
self.bh = self.add_weight(
|
|
40
|
+
shape=(self.hidden_size,),
|
|
41
|
+
initializer="zeros",
|
|
42
|
+
trainable=True,
|
|
43
|
+
name="bh",
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
def call(self, x):
|
|
47
|
+
|
|
48
|
+
batch_size = x.shape[0]
|
|
49
|
+
time_steps = x.shape[1]
|
|
50
|
+
|
|
51
|
+
h = Tensor.zeros(
|
|
52
|
+
(batch_size, self.hidden_size),
|
|
53
|
+
requires_grad=False,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
outputs = []
|
|
57
|
+
|
|
58
|
+
for t in range(time_steps):
|
|
59
|
+
|
|
60
|
+
xt = x[:, t, :]
|
|
61
|
+
|
|
62
|
+
h = Tanh.forward(
|
|
63
|
+
xt @ self.Wx +
|
|
64
|
+
h @ self.Wh +
|
|
65
|
+
self.bh
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
if self.return_sequences:
|
|
69
|
+
outputs.append(h)
|
|
70
|
+
|
|
71
|
+
if self.return_sequences:
|
|
72
|
+
return Tensor.stack(outputs, axis=1)
|
|
73
|
+
|
|
74
|
+
return h
|
|
75
|
+
|
|
76
|
+
def compute_output_shape(self, input_shape):
|
|
77
|
+
|
|
78
|
+
batch = input_shape[0]
|
|
79
|
+
|
|
80
|
+
return (
|
|
81
|
+
batch,
|
|
82
|
+
self.hidden_size,
|
|
83
|
+
)
|
src/neural/__init__.py
ADDED
|
File without changes
|
src/ops/__init__.py
ADDED
|
File without changes
|
src/ops/stack.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
|
|
3
|
+
from src.core.Tensor import Tensor
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Stack:
|
|
7
|
+
|
|
8
|
+
@staticmethod
|
|
9
|
+
def forward(tensors, axis=0):
|
|
10
|
+
|
|
11
|
+
data = np.stack(
|
|
12
|
+
[t.data for t in tensors],
|
|
13
|
+
axis=axis,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
out = Tensor(
|
|
17
|
+
data,
|
|
18
|
+
requires_grad=any(t.requires_grad for t in tensors),
|
|
19
|
+
parents=tuple(tensors),
|
|
20
|
+
op="Stack",
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
def _backward():
|
|
24
|
+
|
|
25
|
+
for i, tensor in enumerate(tensors):
|
|
26
|
+
|
|
27
|
+
if not tensor.requires_grad:
|
|
28
|
+
continue
|
|
29
|
+
|
|
30
|
+
grad = np.take(
|
|
31
|
+
out.grad,
|
|
32
|
+
indices=i,
|
|
33
|
+
axis=axis,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
tensor.grad += grad
|
|
37
|
+
|
|
38
|
+
out._backward = _backward
|
|
39
|
+
|
|
40
|
+
return out
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class Adagrad:
|
|
5
|
+
|
|
6
|
+
def __init__(self, lr=0.01, eps=1e-8):
|
|
7
|
+
self.lr = lr
|
|
8
|
+
self.eps = eps
|
|
9
|
+
self.cache = {}
|
|
10
|
+
|
|
11
|
+
def step(self, params):
|
|
12
|
+
|
|
13
|
+
for p in params:
|
|
14
|
+
|
|
15
|
+
if not p.requires_grad:
|
|
16
|
+
continue
|
|
17
|
+
|
|
18
|
+
if id(p) not in self.cache:
|
|
19
|
+
self.cache[id(p)] = np.zeros_like(p.data)
|
|
20
|
+
|
|
21
|
+
self.cache[id(p)] += p.grad ** 2
|
|
22
|
+
|
|
23
|
+
p.data -= (
|
|
24
|
+
self.lr
|
|
25
|
+
* p.grad
|
|
26
|
+
/ (np.sqrt(self.cache[id(p)]) + self.eps)
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
def zero_grad(self, params):
|
|
30
|
+
for p in params:
|
|
31
|
+
p.zero_grad()
|