x-transformers 1.42.26__py3-none-any.whl → 1.42.28__py3-none-any.whl
Sign up to get free protection for your applications and to get access to all the features.
- x_transformers/x_transformers.py +1 -1
- {x_transformers-1.42.26.dist-info → x_transformers-1.42.28.dist-info}/METADATA +1 -1
- {x_transformers-1.42.26.dist-info → x_transformers-1.42.28.dist-info}/RECORD +6 -6
- {x_transformers-1.42.26.dist-info → x_transformers-1.42.28.dist-info}/LICENSE +0 -0
- {x_transformers-1.42.26.dist-info → x_transformers-1.42.28.dist-info}/WHEEL +0 -0
- {x_transformers-1.42.26.dist-info → x_transformers-1.42.28.dist-info}/top_level.txt +0 -0
x_transformers/x_transformers.py
CHANGED
@@ -1584,7 +1584,7 @@ class AttentionLayers(Module):
|
|
1584
1584
|
unet_skips = False,
|
1585
1585
|
reinject_input = False, # seen first in DEQ paper https://arxiv.org/abs/1909.01377, but later used in a number of papers trying to achieve depthwise generalization https://arxiv.org/abs/2410.03020v1
|
1586
1586
|
add_value_residual = False, # resformer from Zhou et al - https://arxiv.org/abs/2410.17897v1
|
1587
|
-
learned_value_residual_mix =
|
1587
|
+
learned_value_residual_mix = True, # seeing big improvements when the value residual mix value is learned per token - credit goes to @faresobeid for taking the first step with learned scalar mix, then @Blinkdl for taking it a step further with data dependent. here we will use per token learned
|
1588
1588
|
rel_pos_kwargs: dict = dict(),
|
1589
1589
|
**kwargs
|
1590
1590
|
):
|
@@ -6,11 +6,11 @@ x_transformers/dpo.py,sha256=xt4OuOWhU8pN3OKN2LZAaC2NC8iiEnchqqcrPWVqf0o,3521
|
|
6
6
|
x_transformers/multi_input.py,sha256=tCh-fTJDj2ib4SMGtsa-AM8MxKzJAQSwqAXOu3HU2mg,9252
|
7
7
|
x_transformers/neo_mlp.py,sha256=XCNnnop9WLarcxap1kGuYc1x8GHvwkZiDRnXOxSl3Po,3452
|
8
8
|
x_transformers/nonautoregressive_wrapper.py,sha256=2NU58hYMgn-4Jzg3mie-mXb0XH_dCN7fjlzd3K1rLUY,10510
|
9
|
-
x_transformers/x_transformers.py,sha256=
|
9
|
+
x_transformers/x_transformers.py,sha256=X4HegsAtCnaL3MAxu07RkZ5WBMgtdbi0W-2c9bXQxew,96696
|
10
10
|
x_transformers/xl_autoregressive_wrapper.py,sha256=CvZMJ6A6PA-Y_bQAhnORwjJBSl6Vjq2IdW5KTdk8NI8,4195
|
11
11
|
x_transformers/xval.py,sha256=7S00kCuab4tWQa-vf-z-XfzADjVj48MoFIr7VSIvttg,8575
|
12
|
-
x_transformers-1.42.
|
13
|
-
x_transformers-1.42.
|
14
|
-
x_transformers-1.42.
|
15
|
-
x_transformers-1.42.
|
16
|
-
x_transformers-1.42.
|
12
|
+
x_transformers-1.42.28.dist-info/LICENSE,sha256=As9u198X-U-vph5noInuUfqsAG2zX_oXPHDmdjwlPPY,1066
|
13
|
+
x_transformers-1.42.28.dist-info/METADATA,sha256=txhDZvzsfiBEPBUg3Ipszv2cWu9sXyd7hhDz4BGsbfc,739
|
14
|
+
x_transformers-1.42.28.dist-info/WHEEL,sha256=PZUExdf71Ui_so67QXpySuHtCi3-J3wvF4ORK6k_S8U,91
|
15
|
+
x_transformers-1.42.28.dist-info/top_level.txt,sha256=hO6KGpFuGucRNEtRfme4A_rGcM53AKwGP7RVlRIxS5Q,15
|
16
|
+
x_transformers-1.42.28.dist-info/RECORD,,
|
File without changes
|
File without changes
|
File without changes
|