pytensorforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli.py +604 -0
- pytensorforge-0.1.0.dist-info/METADATA +103 -0
- pytensorforge-0.1.0.dist-info/RECORD +146 -0
- pytensorforge-0.1.0.dist-info/WHEEL +5 -0
- pytensorforge-0.1.0.dist-info/entry_points.txt +2 -0
- pytensorforge-0.1.0.dist-info/top_level.txt +2 -0
- src/__init__.py +0 -0
- src/activations/Activation.py +4 -0
- src/activations/ELU.py +11 -0
- src/activations/GELU.py +6 -0
- src/activations/ReLU.py +27 -0
- src/activations/SELU.py +14 -0
- src/activations/Sigmoid.py +27 -0
- src/activations/Softmax.py +84 -0
- src/activations/Tanh.py +29 -0
- src/activations/__init__.py +17 -0
- src/config.py +120 -0
- src/core/Matrix.py +3 -0
- src/core/Scalar.py +18 -0
- src/core/Tensor.py +866 -0
- src/core/Vector.py +31 -0
- src/core/__init__.py +0 -0
- src/data/__init__.py +0 -0
- src/data/chat_dataset.py +188 -0
- src/data/corpus.py +104 -0
- src/data/document_stream.py +178 -0
- src/data/parallel_encode.py +86 -0
- src/data/prefetch.py +62 -0
- src/data/shard_builder.py +119 -0
- src/data/shard_writer.py +81 -0
- src/data/sharded_dataset.py +112 -0
- src/data/streaming_dataset.py +132 -0
- src/data/validation.py +212 -0
- src/inference/__init__.py +0 -0
- src/inference/chat_template.py +384 -0
- src/inference/config.py +48 -0
- src/inference/engine.py +241 -0
- src/inference/export.py +133 -0
- src/inference/kv_cache.py +65 -0
- src/inference/runtime.py +161 -0
- src/inference/sampling.py +42 -0
- src/inference/scheduler.py +473 -0
- src/inference/text.py +67 -0
- src/initializers/Constant.py +9 -0
- src/initializers/GlorotNormal.py +15 -0
- src/initializers/GlorotUniform.py +26 -0
- src/initializers/HeNormal.py +15 -0
- src/initializers/HeUniform.py +14 -0
- src/initializers/Initializer.py +4 -0
- src/initializers/LecunNormal.py +16 -0
- src/initializers/LecunUniform.py +14 -0
- src/initializers/Ones.py +6 -0
- src/initializers/Orthogonal.py +14 -0
- src/initializers/RandomNormal.py +14 -0
- src/initializers/RandomUniform.py +14 -0
- src/initializers/Zeros.py +8 -0
- src/initializers/__init__.py +17 -0
- src/loss/CategoricalCrossEntropy.py +9 -0
- src/loss/CrossEntropyLoss.py +34 -0
- src/loss/CrossEntropyWithLogitsLoss.py +59 -0
- src/loss/Hinge.py +5 -0
- src/loss/Huber.py +22 -0
- src/loss/Loss.py +6 -0
- src/loss/MSE.py +7 -0
- src/loss/MSELoss.py +10 -0
- src/loss/SparseCategoricalCrossEntropy.py +15 -0
- src/loss/__init__.py +18 -0
- src/loss/bce.py +34 -0
- src/loss/mae.py +16 -0
- src/math/__init__.py +0 -0
- src/math/clip.py +37 -0
- src/math/exp.py +27 -0
- src/math/log.py +25 -0
- src/math/sigmoid.py +5 -0
- src/models/__init__.py +0 -0
- src/models/embedding/Embedding.py +65 -0
- src/models/embedding/__init__.py +0 -0
- src/models/gpt/__init__.py +0 -0
- src/models/gpt/attention.py +158 -0
- src/models/gpt/block.py +74 -0
- src/models/gpt/config.py +103 -0
- src/models/gpt/context.py +44 -0
- src/models/gpt/model.py +165 -0
- src/models/gpt/recompute.py +35 -0
- src/models/gpt/rope.py +84 -0
- src/models/regression/Linear.py +51 -0
- src/models/regression/Logistic.py +36 -0
- src/models/regression/__init__.py +0 -0
- src/models/seq/Sequential.py +297 -0
- src/models/seq/__init__.py +0 -0
- src/models/svm/__init__.py +0 -0
- src/models/tokenizer/BPETokenizer.py +228 -0
- src/models/tokenizer/__init__.py +0 -0
- src/models/transformers/Dropout.py +35 -0
- src/models/transformers/LastToken.py +10 -0
- src/models/transformers/LayerNorm.py +54 -0
- src/models/transformers/Linear.py +18 -0
- src/models/transformers/MultiHeadAttention.py +130 -0
- src/models/transformers/TransformerBlock.py +79 -0
- src/models/transformers/__init__.py +0 -0
- src/neural/Dense.py +58 -0
- src/neural/LSTM.py +167 -0
- src/neural/Layer.py +72 -0
- src/neural/Parameter.py +30 -0
- src/neural/RNN.py +83 -0
- src/neural/__init__.py +0 -0
- src/ops/__init__.py +0 -0
- src/ops/stack.py +40 -0
- src/optimizers/Adagrad.py +31 -0
- src/optimizers/Adam.py +98 -0
- src/optimizers/AdamW.py +84 -0
- src/optimizers/Batch.py +11 -0
- src/optimizers/Nesterov.py +35 -0
- src/optimizers/Optimizer.py +18 -0
- src/optimizers/RMSProp.py +35 -0
- src/optimizers/SGD.py +30 -0
- src/optimizers/SGDMomentum.py +28 -0
- src/optimizers/__init__.py +9 -0
- src/scaling/StandardScaler.py +15 -0
- src/scaling/__init__.py +0 -0
- src/serialization/__init__.py +0 -0
- src/serialization/checkpoint.py +58 -0
- src/serialization/modelio.py +132 -0
- src/serving/__init__.py +0 -0
- src/serving/app.py +792 -0
- src/serving/config.py +216 -0
- src/serving/errors.py +51 -0
- src/serving/http.py +599 -0
- src/serving/metrics.py +293 -0
- src/serving/model_server.py +287 -0
- src/serving/protocol.py +377 -0
- src/serving/security.py +200 -0
- src/serving/server.py +121 -0
- src/tokenization/__init__.py +0 -0
- src/tokenization/base.py +75 -0
- src/tokenization/bpe.py +190 -0
- src/tokenization/bytebpe.py +476 -0
- src/tokenization/registry.py +28 -0
- src/training/__init__.py +0 -0
- src/training/checkpoint_manager.py +101 -0
- src/training/experiment.py +71 -0
- src/training/losses.py +42 -0
- src/training/precision.py +141 -0
- src/training/profiler.py +38 -0
- src/training/scheduler.py +50 -0
- src/training/trainer.py +594 -0
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
from src.loss import losses
|
|
3
|
+
from src.optimizers import optimizers
|
|
4
|
+
from src.serialization.checkpoint import Checkpoint
|
|
5
|
+
from src.serialization.modelio import ModelIO, flatten, unflatten
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Sequential:
|
|
9
|
+
|
|
10
|
+
def __init__(self):
|
|
11
|
+
self.layers = []
|
|
12
|
+
|
|
13
|
+
self.loss = None
|
|
14
|
+
self.optimizer = None
|
|
15
|
+
self.metrics = []
|
|
16
|
+
|
|
17
|
+
def add(self, layer):
|
|
18
|
+
self.layers.append(layer)
|
|
19
|
+
|
|
20
|
+
def __call__(self, x):
|
|
21
|
+
for layer in self.layers:
|
|
22
|
+
x = layer(x)
|
|
23
|
+
# print(layer.__class__.__name__, x.shape)
|
|
24
|
+
|
|
25
|
+
return x
|
|
26
|
+
|
|
27
|
+
def compile(
|
|
28
|
+
self,
|
|
29
|
+
loss,
|
|
30
|
+
optimizer,
|
|
31
|
+
metrics=None,
|
|
32
|
+
):
|
|
33
|
+
|
|
34
|
+
if isinstance(loss, str):
|
|
35
|
+
try:
|
|
36
|
+
self.loss = losses[loss]
|
|
37
|
+
except KeyError:
|
|
38
|
+
raise ValueError(f"Unknown loss '{loss}'")
|
|
39
|
+
else:
|
|
40
|
+
self.loss = loss
|
|
41
|
+
|
|
42
|
+
if isinstance(optimizer, str):
|
|
43
|
+
try:
|
|
44
|
+
self.optimizer = optimizers[optimizer]
|
|
45
|
+
except KeyError:
|
|
46
|
+
raise ValueError(f"Unknown optimizer '{optimizer}'")
|
|
47
|
+
else:
|
|
48
|
+
self.optimizer = optimizer
|
|
49
|
+
|
|
50
|
+
self.metrics = metrics or []
|
|
51
|
+
|
|
52
|
+
def fit(
|
|
53
|
+
self,
|
|
54
|
+
X,
|
|
55
|
+
y,
|
|
56
|
+
epochs=5,
|
|
57
|
+
initial_epoch=0,
|
|
58
|
+
batch_size=32,
|
|
59
|
+
verbose=1,
|
|
60
|
+
checkpoint_path=None,
|
|
61
|
+
checkpoint_every=1,
|
|
62
|
+
):
|
|
63
|
+
n = len(X)
|
|
64
|
+
|
|
65
|
+
for epoch in range(initial_epoch, epochs):
|
|
66
|
+
|
|
67
|
+
epoch_loss = 0.0
|
|
68
|
+
|
|
69
|
+
for start in range(0, n, batch_size):
|
|
70
|
+
|
|
71
|
+
end = start + batch_size
|
|
72
|
+
|
|
73
|
+
xb = X[start:end]
|
|
74
|
+
yb = y[start:end]
|
|
75
|
+
|
|
76
|
+
self.optimizer.zero_grad(self.parameters())
|
|
77
|
+
|
|
78
|
+
predictions = self(xb)
|
|
79
|
+
|
|
80
|
+
loss = self.loss(predictions, yb)
|
|
81
|
+
epoch_loss += loss.data.item()
|
|
82
|
+
|
|
83
|
+
loss.backward()
|
|
84
|
+
|
|
85
|
+
self.optimizer.step(self.parameters())
|
|
86
|
+
|
|
87
|
+
if verbose:
|
|
88
|
+
# if epoch % 20 == 0 and self.layers[1].__class__.__name__:
|
|
89
|
+
# print(np.linalg.norm(self.layers[1].attn.Wq.grad))
|
|
90
|
+
# print(np.linalg.norm(self.layers[1].attn.Wk.grad))
|
|
91
|
+
# print(np.linalg.norm(self.layers[1].attn.Wv.grad))
|
|
92
|
+
# print(np.linalg.norm(self.layers[1].attn.Wo.grad))
|
|
93
|
+
|
|
94
|
+
print(f"Epoch {epoch + 1}/{epochs} loss={epoch_loss:.4f}")
|
|
95
|
+
|
|
96
|
+
if checkpoint_path is not None:
|
|
97
|
+
if (epoch + 1) % checkpoint_every == 0:
|
|
98
|
+
Checkpoint.save(
|
|
99
|
+
model=self,
|
|
100
|
+
optimizer=self.optimizer,
|
|
101
|
+
epoch=epoch + 1,
|
|
102
|
+
loss=epoch_loss,
|
|
103
|
+
path=checkpoint_path,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
def parameters(self):
|
|
107
|
+
params = []
|
|
108
|
+
|
|
109
|
+
for layer in self.layers:
|
|
110
|
+
params.extend(layer.parameters())
|
|
111
|
+
|
|
112
|
+
return params
|
|
113
|
+
|
|
114
|
+
def summary(self):
|
|
115
|
+
|
|
116
|
+
print("=" * 80)
|
|
117
|
+
print(f'{"Model: Sequential":^80}')
|
|
118
|
+
print("=" * 80)
|
|
119
|
+
|
|
120
|
+
print(
|
|
121
|
+
f'{"Layer (type)":30}'
|
|
122
|
+
f'{"Output Shape":25}'
|
|
123
|
+
f'{"Param #":>15}'
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
print("-" * 80)
|
|
127
|
+
|
|
128
|
+
total = 0
|
|
129
|
+
trainable = 0
|
|
130
|
+
non_trainable = 0
|
|
131
|
+
|
|
132
|
+
shape = None
|
|
133
|
+
|
|
134
|
+
for layer in self.layers:
|
|
135
|
+
|
|
136
|
+
params = 0
|
|
137
|
+
|
|
138
|
+
for p in layer.parameters():
|
|
139
|
+
count = int(np.prod(p.shape))
|
|
140
|
+
params += count
|
|
141
|
+
total += count
|
|
142
|
+
|
|
143
|
+
if p.trainable:
|
|
144
|
+
trainable += count
|
|
145
|
+
else:
|
|
146
|
+
non_trainable += count
|
|
147
|
+
|
|
148
|
+
output_shape = (
|
|
149
|
+
str(layer.output_shape)
|
|
150
|
+
if hasattr(layer, "output_shape")
|
|
151
|
+
else "?"
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
print(
|
|
155
|
+
f"{layer.__class__.__name__:30}"
|
|
156
|
+
f"{output_shape:25}"
|
|
157
|
+
f"{params:>15,}"
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
print("-" * 80)
|
|
161
|
+
|
|
162
|
+
print(f"{'Total params:':30}{total:>20,}")
|
|
163
|
+
print(f"{'Trainable params:':30}{trainable:>20,}")
|
|
164
|
+
print(f"{'Non-trainable params:':30}{non_trainable:>20,}")
|
|
165
|
+
|
|
166
|
+
print("=" * 80)
|
|
167
|
+
|
|
168
|
+
if self.optimizer is not None:
|
|
169
|
+
print(f"Optimizer : {self.optimizer.__class__.__name__}")
|
|
170
|
+
|
|
171
|
+
if self.loss is not None:
|
|
172
|
+
print(f"Loss : {self.loss.__class__.__name__}")
|
|
173
|
+
|
|
174
|
+
if self.metrics:
|
|
175
|
+
print(
|
|
176
|
+
"Metrics : "
|
|
177
|
+
+ ", ".join(
|
|
178
|
+
metric.__class__.__name__
|
|
179
|
+
if not isinstance(metric, str)
|
|
180
|
+
else metric
|
|
181
|
+
for metric in self.metrics
|
|
182
|
+
)
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
print("=" * 80)
|
|
186
|
+
|
|
187
|
+
def predict(self, X):
|
|
188
|
+
predictions = []
|
|
189
|
+
for layer in self.layers:
|
|
190
|
+
X = layer(X)
|
|
191
|
+
predictions.append(X)
|
|
192
|
+
return X
|
|
193
|
+
|
|
194
|
+
def state_dict(self):
|
|
195
|
+
|
|
196
|
+
state = {}
|
|
197
|
+
|
|
198
|
+
for i, layer in enumerate(self.layers):
|
|
199
|
+
state[f"layer_{i}"] = layer.state_dict()
|
|
200
|
+
|
|
201
|
+
return state
|
|
202
|
+
|
|
203
|
+
def load_state_dict(self, state):
|
|
204
|
+
|
|
205
|
+
for i, layer in enumerate(self.layers):
|
|
206
|
+
layer.load_state_dict(
|
|
207
|
+
state.get(f"layer_{i}", {})
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
def save(self, path, metadata = None):
|
|
211
|
+
ModelIO.save(self, path, metadata)
|
|
212
|
+
|
|
213
|
+
def load_model(self, path, input_shape=None):
|
|
214
|
+
|
|
215
|
+
if not self.built:
|
|
216
|
+
|
|
217
|
+
if input_shape is None:
|
|
218
|
+
raise RuntimeError(
|
|
219
|
+
"Model must be built before loading."
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
self.build(input_shape)
|
|
223
|
+
|
|
224
|
+
return ModelIO.load(self, path)
|
|
225
|
+
|
|
226
|
+
def load(self, path):
|
|
227
|
+
return ModelIO.load(self, path)
|
|
228
|
+
|
|
229
|
+
def save_metadata(self, path, metadata):
|
|
230
|
+
ModelIO.save_data(metadata, path)
|
|
231
|
+
|
|
232
|
+
def load_metadata(self, path):
|
|
233
|
+
return ModelIO.load(path)
|
|
234
|
+
|
|
235
|
+
def save_checkpoint(
|
|
236
|
+
self,
|
|
237
|
+
path,
|
|
238
|
+
optimizer,
|
|
239
|
+
epoch,
|
|
240
|
+
loss,
|
|
241
|
+
metadata=None,
|
|
242
|
+
):
|
|
243
|
+
|
|
244
|
+
checkpoint = {
|
|
245
|
+
"model": self.state_dict(),
|
|
246
|
+
"optimizer": optimizer.state_dict(),
|
|
247
|
+
"epoch": epoch,
|
|
248
|
+
"loss": loss,
|
|
249
|
+
"metadata": metadata or {},
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
np.savez_compressed(
|
|
253
|
+
path,
|
|
254
|
+
**flatten(checkpoint)
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
def load_checkpoint(
|
|
258
|
+
self,
|
|
259
|
+
path,
|
|
260
|
+
optimizer=None,
|
|
261
|
+
):
|
|
262
|
+
|
|
263
|
+
data = np.load(path, allow_pickle=True)
|
|
264
|
+
|
|
265
|
+
flat = {
|
|
266
|
+
k: data[k]
|
|
267
|
+
for k in data.files
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
checkpoint = unflatten(flat)
|
|
271
|
+
|
|
272
|
+
self.load_state_dict(
|
|
273
|
+
checkpoint["model"]
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
if optimizer is not None:
|
|
277
|
+
optimizer.load_state_dict(
|
|
278
|
+
checkpoint["optimizer"]
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
return {
|
|
282
|
+
"epoch": int(checkpoint["epoch"]),
|
|
283
|
+
"loss": float(checkpoint["loss"]),
|
|
284
|
+
"metadata": checkpoint.get("metadata", {}),
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
def build(self, input_shape):
|
|
288
|
+
|
|
289
|
+
shape = input_shape
|
|
290
|
+
|
|
291
|
+
for layer in self.layers:
|
|
292
|
+
layer.build(shape)
|
|
293
|
+
layer.built = True
|
|
294
|
+
|
|
295
|
+
shape = layer.compute_output_shape(shape)
|
|
296
|
+
|
|
297
|
+
self.built = True
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
from collections import Counter
|
|
2
|
+
|
|
3
|
+
# tokenizer.fit(texts)
|
|
4
|
+
#
|
|
5
|
+
# tokenizer.encode(text)
|
|
6
|
+
#
|
|
7
|
+
# tokenizer.decode(ids)
|
|
8
|
+
#
|
|
9
|
+
# tokenizer.batch_encode(texts)
|
|
10
|
+
#
|
|
11
|
+
# tokenizer.batch_decode(batch)
|
|
12
|
+
#
|
|
13
|
+
# tokenizer.save(path)
|
|
14
|
+
#
|
|
15
|
+
# Tokenizer.load(path)
|
|
16
|
+
#
|
|
17
|
+
# len(tokenizer)
|
|
18
|
+
#
|
|
19
|
+
# tokenizer.vocab_size
|
|
20
|
+
|
|
21
|
+
# tokenizer = Tokenizer(
|
|
22
|
+
# lower=True,
|
|
23
|
+
# oov_token="<unk>",
|
|
24
|
+
# )
|
|
25
|
+
#
|
|
26
|
+
# tokenizer.fit(texts)
|
|
27
|
+
#
|
|
28
|
+
# ids = tokenizer.encode("What is Python?")
|
|
29
|
+
# text = tokenizer.decode(ids)
|
|
30
|
+
#
|
|
31
|
+
# batch = tokenizer.batch_encode(texts)
|
|
32
|
+
#
|
|
33
|
+
# tokenizer.save("tokenizer.ptf")
|
|
34
|
+
# tokenizer = Tokenizer.load("tokenizer.ptf")
|
|
35
|
+
|
|
36
|
+
class BPETokenizer:
|
|
37
|
+
|
|
38
|
+
def __init__(self,
|
|
39
|
+
vocab_size=30000,
|
|
40
|
+
lowercase=True):
|
|
41
|
+
|
|
42
|
+
self.vocab_size = vocab_size
|
|
43
|
+
self.lowercase = lowercase
|
|
44
|
+
|
|
45
|
+
self.merges = []
|
|
46
|
+
self.word_to_index = {}
|
|
47
|
+
self.index_to_word = {}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def build_words(self, texts):
|
|
51
|
+
|
|
52
|
+
words = Counter()
|
|
53
|
+
|
|
54
|
+
for text in texts:
|
|
55
|
+
|
|
56
|
+
if self.lowercase:
|
|
57
|
+
text = text.lower()
|
|
58
|
+
|
|
59
|
+
for word in text.split():
|
|
60
|
+
|
|
61
|
+
chars = tuple(word) + ("</w>",)
|
|
62
|
+
|
|
63
|
+
words[chars] += 1
|
|
64
|
+
|
|
65
|
+
return words
|
|
66
|
+
|
|
67
|
+
def get_pair_counts(self, words):
|
|
68
|
+
|
|
69
|
+
pairs = Counter()
|
|
70
|
+
|
|
71
|
+
for word, freq in words.items():
|
|
72
|
+
|
|
73
|
+
for i in range(len(word) - 1):
|
|
74
|
+
pair = (word[i], word[i + 1])
|
|
75
|
+
|
|
76
|
+
pairs[pair] += freq
|
|
77
|
+
|
|
78
|
+
return pairs
|
|
79
|
+
|
|
80
|
+
def merge(self, words, pair):
|
|
81
|
+
|
|
82
|
+
merged = {}
|
|
83
|
+
|
|
84
|
+
bigram = pair
|
|
85
|
+
|
|
86
|
+
replacement = "".join(pair)
|
|
87
|
+
|
|
88
|
+
for word, freq in words.items():
|
|
89
|
+
|
|
90
|
+
new = []
|
|
91
|
+
|
|
92
|
+
i = 0
|
|
93
|
+
|
|
94
|
+
while i < len(word):
|
|
95
|
+
|
|
96
|
+
if (
|
|
97
|
+
i < len(word) - 1
|
|
98
|
+
and word[i] == bigram[0]
|
|
99
|
+
and word[i + 1] == bigram[1]
|
|
100
|
+
):
|
|
101
|
+
|
|
102
|
+
new.append(replacement)
|
|
103
|
+
|
|
104
|
+
i += 2
|
|
105
|
+
|
|
106
|
+
else:
|
|
107
|
+
|
|
108
|
+
new.append(word[i])
|
|
109
|
+
|
|
110
|
+
i += 1
|
|
111
|
+
|
|
112
|
+
merged[tuple(new)] = freq
|
|
113
|
+
|
|
114
|
+
return merged
|
|
115
|
+
|
|
116
|
+
def fit(self, texts):
|
|
117
|
+
|
|
118
|
+
words = self.build_words(texts)
|
|
119
|
+
|
|
120
|
+
while True:
|
|
121
|
+
|
|
122
|
+
pairs = self.get_pair_counts(words)
|
|
123
|
+
|
|
124
|
+
if not pairs:
|
|
125
|
+
break
|
|
126
|
+
|
|
127
|
+
best = max(
|
|
128
|
+
pairs,
|
|
129
|
+
key=pairs.get
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
self.merges.append(best)
|
|
133
|
+
|
|
134
|
+
words = self.merge(words, best)
|
|
135
|
+
|
|
136
|
+
vocab = set()
|
|
137
|
+
|
|
138
|
+
for word in words:
|
|
139
|
+
vocab.update(word)
|
|
140
|
+
|
|
141
|
+
if len(vocab) >= self.vocab_size:
|
|
142
|
+
break
|
|
143
|
+
|
|
144
|
+
vocab = sorted(vocab)
|
|
145
|
+
|
|
146
|
+
self.word_to_index = {
|
|
147
|
+
token: i
|
|
148
|
+
for i, token in enumerate(vocab)
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
self.index_to_word = {
|
|
152
|
+
i: token
|
|
153
|
+
for token, i in self.word_to_index.items()
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
def encode_word(self, word):
|
|
157
|
+
|
|
158
|
+
tokens = list(word)
|
|
159
|
+
|
|
160
|
+
tokens.append("</w>")
|
|
161
|
+
|
|
162
|
+
for pair in self.merges:
|
|
163
|
+
|
|
164
|
+
replacement = "".join(pair)
|
|
165
|
+
|
|
166
|
+
i = 0
|
|
167
|
+
|
|
168
|
+
merged = []
|
|
169
|
+
|
|
170
|
+
while i < len(tokens):
|
|
171
|
+
|
|
172
|
+
if (
|
|
173
|
+
i < len(tokens) - 1
|
|
174
|
+
and tokens[i] == pair[0]
|
|
175
|
+
and tokens[i + 1] == pair[1]
|
|
176
|
+
):
|
|
177
|
+
|
|
178
|
+
merged.append(replacement)
|
|
179
|
+
|
|
180
|
+
i += 2
|
|
181
|
+
|
|
182
|
+
else:
|
|
183
|
+
|
|
184
|
+
merged.append(tokens[i])
|
|
185
|
+
|
|
186
|
+
i += 1
|
|
187
|
+
|
|
188
|
+
tokens = merged
|
|
189
|
+
|
|
190
|
+
return [
|
|
191
|
+
self.word_to_index[t]
|
|
192
|
+
for t in tokens
|
|
193
|
+
]
|
|
194
|
+
|
|
195
|
+
def encode(self, text):
|
|
196
|
+
|
|
197
|
+
if self.lowercase:
|
|
198
|
+
text = text.lower()
|
|
199
|
+
|
|
200
|
+
ids = []
|
|
201
|
+
|
|
202
|
+
for word in text.split():
|
|
203
|
+
ids.extend(self.encode_word(word))
|
|
204
|
+
|
|
205
|
+
return ids
|
|
206
|
+
|
|
207
|
+
def decode(self, ids):
|
|
208
|
+
|
|
209
|
+
tokens = [
|
|
210
|
+
self.index_to_word[i]
|
|
211
|
+
for i in ids
|
|
212
|
+
]
|
|
213
|
+
|
|
214
|
+
text = ""
|
|
215
|
+
|
|
216
|
+
for token in tokens:
|
|
217
|
+
|
|
218
|
+
if token == "</w>":
|
|
219
|
+
|
|
220
|
+
text += " "
|
|
221
|
+
|
|
222
|
+
else:
|
|
223
|
+
|
|
224
|
+
text += token
|
|
225
|
+
|
|
226
|
+
return text.strip()
|
|
227
|
+
|
|
228
|
+
|
|
File without changes
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
|
|
3
|
+
from src.core.Tensor import Tensor
|
|
4
|
+
from src.neural.Layer import Layer
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class Dropout(Layer):
|
|
8
|
+
|
|
9
|
+
def __init__(self, p=0.5):
|
|
10
|
+
super().__init__()
|
|
11
|
+
self.p = p
|
|
12
|
+
self.training = True
|
|
13
|
+
|
|
14
|
+
def build(self, input_shape):
|
|
15
|
+
pass
|
|
16
|
+
|
|
17
|
+
def call(self, x):
|
|
18
|
+
|
|
19
|
+
if not self.training or self.p <= 0:
|
|
20
|
+
return x
|
|
21
|
+
|
|
22
|
+
keep = 1 - self.p
|
|
23
|
+
|
|
24
|
+
mask = np.random.binomial(
|
|
25
|
+
1,
|
|
26
|
+
keep,
|
|
27
|
+
size=x.shape,
|
|
28
|
+
).astype(np.float32)
|
|
29
|
+
|
|
30
|
+
mask = Tensor(
|
|
31
|
+
mask / keep,
|
|
32
|
+
requires_grad=False,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
return x * mask
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
|
|
3
|
+
from src.neural.Layer import Layer
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class LayerNorm(Layer):
|
|
7
|
+
def __init__(self, d_model, eps=1e-6):
|
|
8
|
+
super().__init__()
|
|
9
|
+
self.beta = None
|
|
10
|
+
self.gamma = None
|
|
11
|
+
self.d_model = d_model
|
|
12
|
+
self.eps = eps
|
|
13
|
+
|
|
14
|
+
def call_(self, x, mask=None):
|
|
15
|
+
mean = x.mean(axis=-1, keepdims=True)
|
|
16
|
+
|
|
17
|
+
# var = ((x - mean) ** 2).mean(
|
|
18
|
+
# axis=-1,
|
|
19
|
+
# keepdims=True,
|
|
20
|
+
# )
|
|
21
|
+
diff = x - mean
|
|
22
|
+
var = (diff * diff).mean(axis=-1, keepdims=True)
|
|
23
|
+
|
|
24
|
+
x_hat = (x - mean) / (var + self.eps).sqrt()
|
|
25
|
+
|
|
26
|
+
return self.gamma * x_hat + self.beta
|
|
27
|
+
|
|
28
|
+
def call(self, x, mask=None):
|
|
29
|
+
# mean = x.mean(axis=-1, keepdims=True)
|
|
30
|
+
#
|
|
31
|
+
# var = ((x - mean) ** 2).mean(axis=-1, keepdims=True)
|
|
32
|
+
#
|
|
33
|
+
# y = self.gamma * (x - mean) / (var + self.eps).sqrt() + self.beta
|
|
34
|
+
# return y
|
|
35
|
+
mean = x.mean(axis=-1, keepdims=True)
|
|
36
|
+
|
|
37
|
+
var = ((x - mean) ** 2).mean(axis=-1, keepdims=True)
|
|
38
|
+
|
|
39
|
+
xhat = (x - mean) / (var + self.eps).sqrt()
|
|
40
|
+
|
|
41
|
+
return self.gamma * xhat + self.beta
|
|
42
|
+
|
|
43
|
+
def build(self, input_shape):
|
|
44
|
+
self.gamma = self.add_weight(
|
|
45
|
+
shape=(self.d_model,),
|
|
46
|
+
initializer="ones",
|
|
47
|
+
name="gamma",
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
self.beta = self.add_weight(
|
|
51
|
+
shape=(self.d_model,),
|
|
52
|
+
initializer="zeros",
|
|
53
|
+
name="beta",
|
|
54
|
+
)
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
from src.neural.Dense import Dense
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class Linear(Dense):
|
|
5
|
+
|
|
6
|
+
def __init__(
|
|
7
|
+
self,
|
|
8
|
+
in_features,
|
|
9
|
+
out_features,
|
|
10
|
+
bias=True,
|
|
11
|
+
):
|
|
12
|
+
super().__init__(
|
|
13
|
+
units=out_features,
|
|
14
|
+
activation=None,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
self.in_features = in_features
|
|
18
|
+
self.use_bias = bias
|