autoregressive-diffusion-pytorch 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: autoregressive-diffusion-pytorch
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: Autoregressive Diffusion - Pytorch
5
5
  Project-URL: Homepage, https://pypi.org/project/autoregressive-diffusion-pytorch/
6
6
  Project-URL: Repository, https://github.com/lucidrains/autoregressive-diffusion-pytorch
@@ -36,7 +36,9 @@ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
36
36
  Requires-Python: >=3.8
37
37
  Requires-Dist: einops>=0.8.0
38
38
  Requires-Dist: einx>=0.3.0
39
+ Requires-Dist: ema-pytorch
39
40
  Requires-Dist: torch>=2.0
41
+ Requires-Dist: torchdiffeq
40
42
  Requires-Dist: tqdm
41
43
  Requires-Dist: x-transformers>=1.31.14
42
44
  Provides-Extra: examples
@@ -52,6 +54,8 @@ Implementation of the architecture behind <a href="https://arxiv.org/abs/2406.11
52
54
 
53
55
  Official repository has been released <a href="https://github.com/LTH14/mar">here</a>
54
56
 
57
+ <a href="https://github.com/lucidrains/transfusion-pytorch">Alternative route</a>
58
+
55
59
  <img src="./images/results.96600.png" width="400px"></img>
56
60
 
57
61
  *oxford flowers at 96k steps*
@@ -217,3 +221,25 @@ trainer()
217
221
  url = {https://api.semanticscholar.org/CorpusID:249240415}
218
222
  }
219
223
  ```
224
+
225
+ ```bibtex
226
+ @article{Liu2022FlowSA,
227
+ title = {Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow},
228
+ author = {Xingchao Liu and Chengyue Gong and Qiang Liu},
229
+ journal = {ArXiv},
230
+ year = {2022},
231
+ volume = {abs/2209.03003},
232
+ url = {https://api.semanticscholar.org/CorpusID:252111177}
233
+ }
234
+ ```
235
+
236
+ ```bibtex
237
+ @article{Esser2024ScalingRF,
238
+ title = {Scaling Rectified Flow Transformers for High-Resolution Image Synthesis},
239
+ author = {Patrick Esser and Sumith Kulal and A. Blattmann and Rahim Entezari and Jonas Muller and Harry Saini and Yam Levi and Dominik Lorenz and Axel Sauer and Frederic Boesel and Dustin Podell and Tim Dockhorn and Zion English and Kyle Lacey and Alex Goodwin and Yannik Marek and Robin Rombach},
240
+ journal = {ArXiv},
241
+ year = {2024},
242
+ volume = {abs/2403.03206},
243
+ url = {https://api.semanticscholar.org/CorpusID:268247980}
244
+ }
245
+ ```
@@ -6,6 +6,8 @@ Implementation of the architecture behind <a href="https://arxiv.org/abs/2406.11
6
6
 
7
7
  Official repository has been released <a href="https://github.com/LTH14/mar">here</a>
8
8
 
9
+ <a href="https://github.com/lucidrains/transfusion-pytorch">Alternative route</a>
10
+
9
11
  <img src="./images/results.96600.png" width="400px"></img>
10
12
 
11
13
  *oxford flowers at 96k steps*
@@ -171,3 +173,25 @@ trainer()
171
173
  url = {https://api.semanticscholar.org/CorpusID:249240415}
172
174
  }
173
175
  ```
176
+
177
+ ```bibtex
178
+ @article{Liu2022FlowSA,
179
+ title = {Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow},
180
+ author = {Xingchao Liu and Chengyue Gong and Qiang Liu},
181
+ journal = {ArXiv},
182
+ year = {2022},
183
+ volume = {abs/2209.03003},
184
+ url = {https://api.semanticscholar.org/CorpusID:252111177}
185
+ }
186
+ ```
187
+
188
+ ```bibtex
189
+ @article{Esser2024ScalingRF,
190
+ title = {Scaling Rectified Flow Transformers for High-Resolution Image Synthesis},
191
+ author = {Patrick Esser and Sumith Kulal and A. Blattmann and Rahim Entezari and Jonas Muller and Harry Saini and Yam Levi and Dominik Lorenz and Axel Sauer and Frederic Boesel and Dustin Podell and Tim Dockhorn and Zion English and Kyle Lacey and Alex Goodwin and Yannik Marek and Robin Rombach},
192
+ journal = {ArXiv},
193
+ year = {2024},
194
+ volume = {abs/2403.03206},
195
+ url = {https://api.semanticscholar.org/CorpusID:268247980}
196
+ }
197
+ ```
@@ -260,7 +260,8 @@ class AutoregressiveFlow(Module):
260
260
 
261
261
  def forward(
262
262
  self,
263
- seq
263
+ seq,
264
+ noised_seq = None
264
265
  ):
265
266
  b, seq_len, dim = seq.shape
266
267
 
@@ -271,6 +272,9 @@ class AutoregressiveFlow(Module):
271
272
 
272
273
  seq, target = seq[:, :-1], seq
273
274
 
275
+ if exists(noised_seq):
276
+ seq = noised_seq[:, :-1]
277
+
274
278
  # append start tokens
275
279
 
276
280
  seq = self.proj_in(seq)
@@ -303,6 +307,7 @@ class ImageAutoregressiveFlow(Module):
303
307
  image_size,
304
308
  patch_size,
305
309
  channels = 3,
310
+ train_max_noise = 0.,
306
311
  model: dict = dict(),
307
312
  ):
308
313
  super().__init__()
@@ -314,6 +319,10 @@ class ImageAutoregressiveFlow(Module):
314
319
  self.image_size = image_size
315
320
  self.patch_size = patch_size
316
321
 
322
+ assert 0. <= train_max_noise < 1.
323
+
324
+ self.train_max_noise = train_max_noise
325
+
317
326
  self.to_tokens = Rearrange('b c (h p1) (w p2) -> b (h w) (c p1 p2)', p1 = patch_size, p2 = patch_size)
318
327
 
319
328
  self.model = AutoregressiveFlow(
@@ -330,6 +339,20 @@ class ImageAutoregressiveFlow(Module):
330
339
  return unnormalize_to_zero_to_one(images)
331
340
 
332
341
  def forward(self, images):
342
+ train_under_noise, device = self.train_max_noise > 0., images.device
343
+
333
344
  images = normalize_to_neg_one_to_one(images)
334
345
  tokens = self.to_tokens(images)
335
- return self.model(tokens)
346
+
347
+ if not train_under_noise:
348
+ return self.model(tokens)
349
+
350
+ # allow for the network to predict from slightly noised images of the past
351
+
352
+ times = torch.rand(images.shape[0], device = device) * self.train_max_noise
353
+ noise = torch.randn_like(images)
354
+ padded_times = right_pad_dims_to(images, times)
355
+ noised_images = images * (1. - padded_times) + noise * padded_times
356
+ noised_tokens = self.to_tokens(noised_images)
357
+
358
+ return self.model(tokens, noised_tokens)
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "autoregressive-diffusion-pytorch"
3
- version = "0.2.2"
3
+ version = "0.2.4"
4
4
  description = "Autoregressive Diffusion - Pytorch"
5
5
  authors = [
6
6
  { name = "Phil Wang", email = "lucidrains@gmail.com" }
@@ -25,8 +25,10 @@ classifiers=[
25
25
  dependencies = [
26
26
  'einx>=0.3.0',
27
27
  'einops>=0.8.0',
28
+ 'ema-pytorch',
28
29
  'x-transformers>=1.31.14',
29
30
  'torch>=2.0',
31
+ 'torchdiffeq',
30
32
  'tqdm'
31
33
  ]
32
34