PyPI - diffusers - Versions diffs - 0.27.1__py3-none-any.whl → 0.32.2__py3-none-any.whl - Mend

diffusers 0.27.1py3-none-any.whl → 0.32.2py3-none-any.whl

Files changed (445) hide show

diffusers/__init__.py +233 -6
diffusers/callbacks.py +209 -0
diffusers/commands/env.py +102 -6
diffusers/configuration_utils.py +45 -16
diffusers/dependency_versions_table.py +4 -3
diffusers/image_processor.py +434 -110
diffusers/loaders/__init__.py +42 -9
diffusers/loaders/ip_adapter.py +626 -36
diffusers/loaders/lora_base.py +900 -0
diffusers/loaders/lora_conversion_utils.py +991 -125
diffusers/loaders/lora_pipeline.py +3812 -0
diffusers/loaders/peft.py +571 -7
diffusers/loaders/single_file.py +405 -173
diffusers/loaders/single_file_model.py +385 -0
diffusers/loaders/single_file_utils.py +1783 -713
diffusers/loaders/textual_inversion.py +41 -23
diffusers/loaders/transformer_flux.py +181 -0
diffusers/loaders/transformer_sd3.py +89 -0
diffusers/loaders/unet.py +464 -540
diffusers/loaders/unet_loader_utils.py +163 -0
diffusers/models/__init__.py +76 -7
diffusers/models/activations.py +65 -10
diffusers/models/adapter.py +53 -53
diffusers/models/attention.py +605 -18
diffusers/models/attention_flax.py +1 -1
diffusers/models/attention_processor.py +4304 -687
diffusers/models/autoencoders/__init__.py +8 -0
diffusers/models/autoencoders/autoencoder_asym_kl.py +15 -17
diffusers/models/autoencoders/autoencoder_dc.py +620 -0
diffusers/models/autoencoders/autoencoder_kl.py +110 -28
diffusers/models/autoencoders/autoencoder_kl_allegro.py +1149 -0
diffusers/models/autoencoders/autoencoder_kl_cogvideox.py +1482 -0
diffusers/models/autoencoders/autoencoder_kl_hunyuan_video.py +1176 -0
diffusers/models/autoencoders/autoencoder_kl_ltx.py +1338 -0
diffusers/models/autoencoders/autoencoder_kl_mochi.py +1166 -0
diffusers/models/autoencoders/autoencoder_kl_temporal_decoder.py +19 -24
diffusers/models/autoencoders/autoencoder_oobleck.py +464 -0
diffusers/models/autoencoders/autoencoder_tiny.py +21 -18
diffusers/models/autoencoders/consistency_decoder_vae.py +45 -20
diffusers/models/autoencoders/vae.py +41 -29
diffusers/models/autoencoders/vq_model.py +182 -0
diffusers/models/controlnet.py +47 -800
diffusers/models/controlnet_flux.py +70 -0
diffusers/models/controlnet_sd3.py +68 -0
diffusers/models/controlnet_sparsectrl.py +116 -0
diffusers/models/controlnets/__init__.py +23 -0
diffusers/models/controlnets/controlnet.py +872 -0
diffusers/models/{controlnet_flax.py → controlnets/controlnet_flax.py} +9 -9
diffusers/models/controlnets/controlnet_flux.py +536 -0
diffusers/models/controlnets/controlnet_hunyuan.py +401 -0
diffusers/models/controlnets/controlnet_sd3.py +489 -0
diffusers/models/controlnets/controlnet_sparsectrl.py +788 -0
diffusers/models/controlnets/controlnet_union.py +832 -0
diffusers/models/controlnets/controlnet_xs.py +1946 -0
diffusers/models/controlnets/multicontrolnet.py +183 -0
diffusers/models/downsampling.py +85 -18
diffusers/models/embeddings.py +1856 -158
diffusers/models/embeddings_flax.py +23 -9
diffusers/models/model_loading_utils.py +480 -0
diffusers/models/modeling_flax_pytorch_utils.py +2 -1
diffusers/models/modeling_flax_utils.py +2 -7
diffusers/models/modeling_outputs.py +14 -0
diffusers/models/modeling_pytorch_flax_utils.py +1 -1
diffusers/models/modeling_utils.py +611 -146
diffusers/models/normalization.py +361 -20
diffusers/models/resnet.py +18 -23
diffusers/models/transformers/__init__.py +16 -0
diffusers/models/transformers/auraflow_transformer_2d.py +544 -0
diffusers/models/transformers/cogvideox_transformer_3d.py +542 -0
diffusers/models/transformers/dit_transformer_2d.py +240 -0
diffusers/models/transformers/dual_transformer_2d.py +9 -8
diffusers/models/transformers/hunyuan_transformer_2d.py +578 -0
diffusers/models/transformers/latte_transformer_3d.py +327 -0
diffusers/models/transformers/lumina_nextdit2d.py +340 -0
diffusers/models/transformers/pixart_transformer_2d.py +445 -0
diffusers/models/transformers/prior_transformer.py +13 -13
diffusers/models/transformers/sana_transformer.py +488 -0
diffusers/models/transformers/stable_audio_transformer.py +458 -0
diffusers/models/transformers/t5_film_transformer.py +17 -19
diffusers/models/transformers/transformer_2d.py +297 -187
diffusers/models/transformers/transformer_allegro.py +422 -0
diffusers/models/transformers/transformer_cogview3plus.py +386 -0
diffusers/models/transformers/transformer_flux.py +593 -0
diffusers/models/transformers/transformer_hunyuan_video.py +791 -0
diffusers/models/transformers/transformer_ltx.py +469 -0
diffusers/models/transformers/transformer_mochi.py +499 -0
diffusers/models/transformers/transformer_sd3.py +461 -0
diffusers/models/transformers/transformer_temporal.py +21 -19
diffusers/models/unets/unet_1d.py +8 -8
diffusers/models/unets/unet_1d_blocks.py +31 -31
diffusers/models/unets/unet_2d.py +17 -10
diffusers/models/unets/unet_2d_blocks.py +225 -149
diffusers/models/unets/unet_2d_condition.py +41 -40
diffusers/models/unets/unet_2d_condition_flax.py +6 -5
diffusers/models/unets/unet_3d_blocks.py +192 -1057
diffusers/models/unets/unet_3d_condition.py +22 -27
diffusers/models/unets/unet_i2vgen_xl.py +22 -18
diffusers/models/unets/unet_kandinsky3.py +2 -2
diffusers/models/unets/unet_motion_model.py +1413 -89
diffusers/models/unets/unet_spatio_temporal_condition.py +40 -16
diffusers/models/unets/unet_stable_cascade.py +19 -18
diffusers/models/unets/uvit_2d.py +2 -2
diffusers/models/upsampling.py +95 -26
diffusers/models/vq_model.py +12 -164
diffusers/optimization.py +1 -1
diffusers/pipelines/__init__.py +202 -3
diffusers/pipelines/allegro/__init__.py +48 -0
diffusers/pipelines/allegro/pipeline_allegro.py +938 -0
diffusers/pipelines/allegro/pipeline_output.py +23 -0
diffusers/pipelines/amused/pipeline_amused.py +12 -12
diffusers/pipelines/amused/pipeline_amused_img2img.py +14 -12
diffusers/pipelines/amused/pipeline_amused_inpaint.py +13 -11
diffusers/pipelines/animatediff/__init__.py +8 -0
diffusers/pipelines/animatediff/pipeline_animatediff.py +122 -109
diffusers/pipelines/animatediff/pipeline_animatediff_controlnet.py +1106 -0
diffusers/pipelines/animatediff/pipeline_animatediff_sdxl.py +1288 -0
diffusers/pipelines/animatediff/pipeline_animatediff_sparsectrl.py +1010 -0
diffusers/pipelines/animatediff/pipeline_animatediff_video2video.py +236 -180
diffusers/pipelines/animatediff/pipeline_animatediff_video2video_controlnet.py +1341 -0
diffusers/pipelines/animatediff/pipeline_output.py +3 -2
diffusers/pipelines/audioldm/pipeline_audioldm.py +14 -14
diffusers/pipelines/audioldm2/modeling_audioldm2.py +58 -39
diffusers/pipelines/audioldm2/pipeline_audioldm2.py +121 -36
diffusers/pipelines/aura_flow/__init__.py +48 -0
diffusers/pipelines/aura_flow/pipeline_aura_flow.py +584 -0
diffusers/pipelines/auto_pipeline.py +196 -28
diffusers/pipelines/blip_diffusion/blip_image_processing.py +1 -1
diffusers/pipelines/blip_diffusion/modeling_blip2.py +6 -6
diffusers/pipelines/blip_diffusion/modeling_ctx_clip.py +1 -1
diffusers/pipelines/blip_diffusion/pipeline_blip_diffusion.py +2 -2
diffusers/pipelines/cogvideo/__init__.py +54 -0
diffusers/pipelines/cogvideo/pipeline_cogvideox.py +772 -0
diffusers/pipelines/cogvideo/pipeline_cogvideox_fun_control.py +825 -0
diffusers/pipelines/cogvideo/pipeline_cogvideox_image2video.py +885 -0
diffusers/pipelines/cogvideo/pipeline_cogvideox_video2video.py +851 -0
diffusers/pipelines/cogvideo/pipeline_output.py +20 -0
diffusers/pipelines/cogview3/__init__.py +47 -0
diffusers/pipelines/cogview3/pipeline_cogview3plus.py +674 -0
diffusers/pipelines/cogview3/pipeline_output.py +21 -0
diffusers/pipelines/consistency_models/pipeline_consistency_models.py +6 -6
diffusers/pipelines/controlnet/__init__.py +86 -80
diffusers/pipelines/controlnet/multicontrolnet.py +7 -182
diffusers/pipelines/controlnet/pipeline_controlnet.py +134 -87
diffusers/pipelines/controlnet/pipeline_controlnet_blip_diffusion.py +2 -2
diffusers/pipelines/controlnet/pipeline_controlnet_img2img.py +93 -77
diffusers/pipelines/controlnet/pipeline_controlnet_inpaint.py +88 -197
diffusers/pipelines/controlnet/pipeline_controlnet_inpaint_sd_xl.py +136 -90
diffusers/pipelines/controlnet/pipeline_controlnet_sd_xl.py +176 -80
diffusers/pipelines/controlnet/pipeline_controlnet_sd_xl_img2img.py +125 -89
diffusers/pipelines/controlnet/pipeline_controlnet_union_inpaint_sd_xl.py +1790 -0
diffusers/pipelines/controlnet/pipeline_controlnet_union_sd_xl.py +1501 -0
diffusers/pipelines/controlnet/pipeline_controlnet_union_sd_xl_img2img.py +1627 -0
diffusers/pipelines/controlnet/pipeline_flax_controlnet.py +2 -2
diffusers/pipelines/controlnet_hunyuandit/__init__.py +48 -0
diffusers/pipelines/controlnet_hunyuandit/pipeline_hunyuandit_controlnet.py +1060 -0
diffusers/pipelines/controlnet_sd3/__init__.py +57 -0
diffusers/pipelines/controlnet_sd3/pipeline_stable_diffusion_3_controlnet.py +1133 -0
diffusers/pipelines/controlnet_sd3/pipeline_stable_diffusion_3_controlnet_inpainting.py +1153 -0
diffusers/pipelines/controlnet_xs/__init__.py +68 -0
diffusers/pipelines/controlnet_xs/pipeline_controlnet_xs.py +916 -0
diffusers/pipelines/controlnet_xs/pipeline_controlnet_xs_sd_xl.py +1111 -0
diffusers/pipelines/ddpm/pipeline_ddpm.py +2 -2
diffusers/pipelines/deepfloyd_if/pipeline_if.py +16 -30
diffusers/pipelines/deepfloyd_if/pipeline_if_img2img.py +20 -35
diffusers/pipelines/deepfloyd_if/pipeline_if_img2img_superresolution.py +23 -41
diffusers/pipelines/deepfloyd_if/pipeline_if_inpainting.py +22 -38
diffusers/pipelines/deepfloyd_if/pipeline_if_inpainting_superresolution.py +25 -41
diffusers/pipelines/deepfloyd_if/pipeline_if_superresolution.py +19 -34
diffusers/pipelines/deepfloyd_if/pipeline_output.py +6 -5
diffusers/pipelines/deepfloyd_if/watermark.py +1 -1
diffusers/pipelines/deprecated/alt_diffusion/modeling_roberta_series.py +11 -11
diffusers/pipelines/deprecated/alt_diffusion/pipeline_alt_diffusion.py +70 -30
diffusers/pipelines/deprecated/alt_diffusion/pipeline_alt_diffusion_img2img.py +48 -25
diffusers/pipelines/deprecated/repaint/pipeline_repaint.py +2 -2
diffusers/pipelines/deprecated/spectrogram_diffusion/pipeline_spectrogram_diffusion.py +7 -7
diffusers/pipelines/deprecated/stable_diffusion_variants/pipeline_cycle_diffusion.py +21 -20
diffusers/pipelines/deprecated/stable_diffusion_variants/pipeline_stable_diffusion_inpaint_legacy.py +27 -29
diffusers/pipelines/deprecated/stable_diffusion_variants/pipeline_stable_diffusion_model_editing.py +33 -27
diffusers/pipelines/deprecated/stable_diffusion_variants/pipeline_stable_diffusion_paradigms.py +33 -23
diffusers/pipelines/deprecated/stable_diffusion_variants/pipeline_stable_diffusion_pix2pix_zero.py +36 -30
diffusers/pipelines/deprecated/versatile_diffusion/modeling_text_unet.py +102 -69
diffusers/pipelines/deprecated/versatile_diffusion/pipeline_versatile_diffusion.py +13 -13
diffusers/pipelines/deprecated/versatile_diffusion/pipeline_versatile_diffusion_dual_guided.py +10 -5
diffusers/pipelines/deprecated/versatile_diffusion/pipeline_versatile_diffusion_image_variation.py +11 -6
diffusers/pipelines/deprecated/versatile_diffusion/pipeline_versatile_diffusion_text_to_image.py +10 -5
diffusers/pipelines/deprecated/vq_diffusion/pipeline_vq_diffusion.py +5 -5
diffusers/pipelines/dit/pipeline_dit.py +7 -4
diffusers/pipelines/flux/__init__.py +69 -0
diffusers/pipelines/flux/modeling_flux.py +47 -0
diffusers/pipelines/flux/pipeline_flux.py +957 -0
diffusers/pipelines/flux/pipeline_flux_control.py +889 -0
diffusers/pipelines/flux/pipeline_flux_control_img2img.py +945 -0
diffusers/pipelines/flux/pipeline_flux_control_inpaint.py +1141 -0
diffusers/pipelines/flux/pipeline_flux_controlnet.py +1006 -0
diffusers/pipelines/flux/pipeline_flux_controlnet_image_to_image.py +998 -0
diffusers/pipelines/flux/pipeline_flux_controlnet_inpainting.py +1204 -0
diffusers/pipelines/flux/pipeline_flux_fill.py +969 -0
diffusers/pipelines/flux/pipeline_flux_img2img.py +856 -0
diffusers/pipelines/flux/pipeline_flux_inpaint.py +1022 -0
diffusers/pipelines/flux/pipeline_flux_prior_redux.py +492 -0
diffusers/pipelines/flux/pipeline_output.py +37 -0
diffusers/pipelines/free_init_utils.py +41 -38
diffusers/pipelines/free_noise_utils.py +596 -0
diffusers/pipelines/hunyuan_video/__init__.py +48 -0
diffusers/pipelines/hunyuan_video/pipeline_hunyuan_video.py +687 -0
diffusers/pipelines/hunyuan_video/pipeline_output.py +20 -0
diffusers/pipelines/hunyuandit/__init__.py +48 -0
diffusers/pipelines/hunyuandit/pipeline_hunyuandit.py +916 -0
diffusers/pipelines/i2vgen_xl/pipeline_i2vgen_xl.py +33 -48
diffusers/pipelines/kandinsky/pipeline_kandinsky.py +8 -8
diffusers/pipelines/kandinsky/pipeline_kandinsky_combined.py +32 -29
diffusers/pipelines/kandinsky/pipeline_kandinsky_img2img.py +11 -11
diffusers/pipelines/kandinsky/pipeline_kandinsky_inpaint.py +12 -12
diffusers/pipelines/kandinsky/pipeline_kandinsky_prior.py +10 -10
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2.py +6 -6
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2_combined.py +34 -31
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2_controlnet.py +10 -10
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2_controlnet_img2img.py +10 -10
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2_img2img.py +6 -6
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2_inpainting.py +8 -8
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2_prior.py +7 -7
diffusers/pipelines/kandinsky2_2/pipeline_kandinsky2_2_prior_emb2emb.py +6 -6
diffusers/pipelines/kandinsky3/convert_kandinsky3_unet.py +3 -3
diffusers/pipelines/kandinsky3/pipeline_kandinsky3.py +22 -35
diffusers/pipelines/kandinsky3/pipeline_kandinsky3_img2img.py +26 -37
diffusers/pipelines/kolors/__init__.py +54 -0
diffusers/pipelines/kolors/pipeline_kolors.py +1070 -0
diffusers/pipelines/kolors/pipeline_kolors_img2img.py +1250 -0
diffusers/pipelines/kolors/pipeline_output.py +21 -0
diffusers/pipelines/kolors/text_encoder.py +889 -0
diffusers/pipelines/kolors/tokenizer.py +338 -0
diffusers/pipelines/latent_consistency_models/pipeline_latent_consistency_img2img.py +82 -62
diffusers/pipelines/latent_consistency_models/pipeline_latent_consistency_text2img.py +77 -60
diffusers/pipelines/latent_diffusion/pipeline_latent_diffusion.py +12 -12
diffusers/pipelines/latte/__init__.py +48 -0
diffusers/pipelines/latte/pipeline_latte.py +881 -0
diffusers/pipelines/ledits_pp/pipeline_leditspp_stable_diffusion.py +80 -74
diffusers/pipelines/ledits_pp/pipeline_leditspp_stable_diffusion_xl.py +85 -76
diffusers/pipelines/ledits_pp/pipeline_output.py +2 -2
diffusers/pipelines/ltx/__init__.py +50 -0
diffusers/pipelines/ltx/pipeline_ltx.py +789 -0
diffusers/pipelines/ltx/pipeline_ltx_image2video.py +885 -0
diffusers/pipelines/ltx/pipeline_output.py +20 -0
diffusers/pipelines/lumina/__init__.py +48 -0
diffusers/pipelines/lumina/pipeline_lumina.py +890 -0
diffusers/pipelines/marigold/__init__.py +50 -0
diffusers/pipelines/marigold/marigold_image_processing.py +576 -0
diffusers/pipelines/marigold/pipeline_marigold_depth.py +813 -0
diffusers/pipelines/marigold/pipeline_marigold_normals.py +690 -0
diffusers/pipelines/mochi/__init__.py +48 -0
diffusers/pipelines/mochi/pipeline_mochi.py +748 -0
diffusers/pipelines/mochi/pipeline_output.py +20 -0
diffusers/pipelines/musicldm/pipeline_musicldm.py +14 -14
diffusers/pipelines/pag/__init__.py +80 -0
diffusers/pipelines/pag/pag_utils.py +243 -0
diffusers/pipelines/pag/pipeline_pag_controlnet_sd.py +1328 -0
diffusers/pipelines/pag/pipeline_pag_controlnet_sd_inpaint.py +1543 -0
diffusers/pipelines/pag/pipeline_pag_controlnet_sd_xl.py +1610 -0
diffusers/pipelines/pag/pipeline_pag_controlnet_sd_xl_img2img.py +1683 -0
diffusers/pipelines/pag/pipeline_pag_hunyuandit.py +969 -0
diffusers/pipelines/pag/pipeline_pag_kolors.py +1136 -0
diffusers/pipelines/pag/pipeline_pag_pixart_sigma.py +865 -0
diffusers/pipelines/pag/pipeline_pag_sana.py +886 -0
diffusers/pipelines/pag/pipeline_pag_sd.py +1062 -0
diffusers/pipelines/pag/pipeline_pag_sd_3.py +994 -0
diffusers/pipelines/pag/pipeline_pag_sd_3_img2img.py +1058 -0
diffusers/pipelines/pag/pipeline_pag_sd_animatediff.py +866 -0
diffusers/pipelines/pag/pipeline_pag_sd_img2img.py +1094 -0
diffusers/pipelines/pag/pipeline_pag_sd_inpaint.py +1356 -0
diffusers/pipelines/pag/pipeline_pag_sd_xl.py +1345 -0
diffusers/pipelines/pag/pipeline_pag_sd_xl_img2img.py +1544 -0
diffusers/pipelines/pag/pipeline_pag_sd_xl_inpaint.py +1776 -0
diffusers/pipelines/paint_by_example/pipeline_paint_by_example.py +17 -12
diffusers/pipelines/pia/pipeline_pia.py +74 -164
diffusers/pipelines/pipeline_flax_utils.py +5 -10
diffusers/pipelines/pipeline_loading_utils.py +515 -53
diffusers/pipelines/pipeline_utils.py +411 -222
diffusers/pipelines/pixart_alpha/__init__.py +8 -1
diffusers/pipelines/pixart_alpha/pipeline_pixart_alpha.py +76 -93
diffusers/pipelines/pixart_alpha/pipeline_pixart_sigma.py +873 -0
diffusers/pipelines/sana/__init__.py +47 -0
diffusers/pipelines/sana/pipeline_output.py +21 -0
diffusers/pipelines/sana/pipeline_sana.py +884 -0
diffusers/pipelines/semantic_stable_diffusion/pipeline_semantic_stable_diffusion.py +27 -23
diffusers/pipelines/shap_e/pipeline_shap_e.py +3 -3
diffusers/pipelines/shap_e/pipeline_shap_e_img2img.py +14 -14
diffusers/pipelines/shap_e/renderer.py +1 -1
diffusers/pipelines/stable_audio/__init__.py +50 -0
diffusers/pipelines/stable_audio/modeling_stable_audio.py +158 -0
diffusers/pipelines/stable_audio/pipeline_stable_audio.py +756 -0
diffusers/pipelines/stable_cascade/pipeline_stable_cascade.py +71 -25
diffusers/pipelines/stable_cascade/pipeline_stable_cascade_combined.py +23 -19
diffusers/pipelines/stable_cascade/pipeline_stable_cascade_prior.py +35 -34
diffusers/pipelines/stable_diffusion/__init__.py +0 -1
diffusers/pipelines/stable_diffusion/convert_from_ckpt.py +20 -11
diffusers/pipelines/stable_diffusion/pipeline_flax_stable_diffusion.py +1 -1
diffusers/pipelines/stable_diffusion/pipeline_onnx_stable_diffusion.py +2 -2
diffusers/pipelines/stable_diffusion/pipeline_onnx_stable_diffusion_upscale.py +6 -6
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion.py +145 -79
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion_depth2img.py +43 -28
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion_image_variation.py +13 -8
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion_img2img.py +100 -68
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion_inpaint.py +109 -201
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion_instruct_pix2pix.py +131 -32
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion_latent_upscale.py +247 -87
diffusers/pipelines/stable_diffusion/pipeline_stable_diffusion_upscale.py +30 -29
diffusers/pipelines/stable_diffusion/pipeline_stable_unclip.py +35 -27
diffusers/pipelines/stable_diffusion/pipeline_stable_unclip_img2img.py +49 -42
diffusers/pipelines/stable_diffusion/safety_checker.py +2 -1
diffusers/pipelines/stable_diffusion_3/__init__.py +54 -0
diffusers/pipelines/stable_diffusion_3/pipeline_output.py +21 -0
diffusers/pipelines/stable_diffusion_3/pipeline_stable_diffusion_3.py +1140 -0
diffusers/pipelines/stable_diffusion_3/pipeline_stable_diffusion_3_img2img.py +1036 -0
diffusers/pipelines/stable_diffusion_3/pipeline_stable_diffusion_3_inpaint.py +1250 -0
diffusers/pipelines/stable_diffusion_attend_and_excite/pipeline_stable_diffusion_attend_and_excite.py +29 -20
diffusers/pipelines/stable_diffusion_diffedit/pipeline_stable_diffusion_diffedit.py +59 -58
diffusers/pipelines/stable_diffusion_gligen/pipeline_stable_diffusion_gligen.py +31 -25
diffusers/pipelines/stable_diffusion_gligen/pipeline_stable_diffusion_gligen_text_image.py +38 -22
diffusers/pipelines/stable_diffusion_k_diffusion/pipeline_stable_diffusion_k_diffusion.py +30 -24
diffusers/pipelines/stable_diffusion_k_diffusion/pipeline_stable_diffusion_xl_k_diffusion.py +24 -23
diffusers/pipelines/stable_diffusion_ldm3d/pipeline_stable_diffusion_ldm3d.py +107 -67
diffusers/pipelines/stable_diffusion_panorama/pipeline_stable_diffusion_panorama.py +316 -69
diffusers/pipelines/stable_diffusion_safe/pipeline_stable_diffusion_safe.py +10 -5
diffusers/pipelines/stable_diffusion_safe/safety_checker.py +1 -1
diffusers/pipelines/stable_diffusion_sag/pipeline_stable_diffusion_sag.py +98 -30
diffusers/pipelines/stable_diffusion_xl/pipeline_stable_diffusion_xl.py +121 -83
diffusers/pipelines/stable_diffusion_xl/pipeline_stable_diffusion_xl_img2img.py +161 -105
diffusers/pipelines/stable_diffusion_xl/pipeline_stable_diffusion_xl_inpaint.py +142 -218
diffusers/pipelines/stable_diffusion_xl/pipeline_stable_diffusion_xl_instruct_pix2pix.py +45 -29
diffusers/pipelines/stable_diffusion_xl/watermark.py +9 -3
diffusers/pipelines/stable_video_diffusion/pipeline_stable_video_diffusion.py +110 -57
diffusers/pipelines/t2i_adapter/pipeline_stable_diffusion_adapter.py +69 -39
diffusers/pipelines/t2i_adapter/pipeline_stable_diffusion_xl_adapter.py +105 -74
diffusers/pipelines/text_to_video_synthesis/pipeline_output.py +3 -2
diffusers/pipelines/text_to_video_synthesis/pipeline_text_to_video_synth.py +29 -49
diffusers/pipelines/text_to_video_synthesis/pipeline_text_to_video_synth_img2img.py +32 -93
diffusers/pipelines/text_to_video_synthesis/pipeline_text_to_video_zero.py +37 -25
diffusers/pipelines/text_to_video_synthesis/pipeline_text_to_video_zero_sdxl.py +54 -40
diffusers/pipelines/unclip/pipeline_unclip.py +6 -6
diffusers/pipelines/unclip/pipeline_unclip_image_variation.py +6 -6
diffusers/pipelines/unidiffuser/modeling_text_decoder.py +1 -1
diffusers/pipelines/unidiffuser/modeling_uvit.py +12 -12
diffusers/pipelines/unidiffuser/pipeline_unidiffuser.py +29 -28
diffusers/pipelines/wuerstchen/modeling_paella_vq_model.py +5 -5
diffusers/pipelines/wuerstchen/modeling_wuerstchen_common.py +5 -10
diffusers/pipelines/wuerstchen/modeling_wuerstchen_prior.py +6 -8
diffusers/pipelines/wuerstchen/pipeline_wuerstchen.py +4 -4
diffusers/pipelines/wuerstchen/pipeline_wuerstchen_combined.py +12 -12
diffusers/pipelines/wuerstchen/pipeline_wuerstchen_prior.py +15 -14
diffusers/{models/dual_transformer_2d.py → quantizers/__init__.py} +2 -6
diffusers/quantizers/auto.py +139 -0
diffusers/quantizers/base.py +233 -0
diffusers/quantizers/bitsandbytes/__init__.py +2 -0
diffusers/quantizers/bitsandbytes/bnb_quantizer.py +561 -0
diffusers/quantizers/bitsandbytes/utils.py +306 -0
diffusers/quantizers/gguf/__init__.py +1 -0
diffusers/quantizers/gguf/gguf_quantizer.py +159 -0
diffusers/quantizers/gguf/utils.py +456 -0
diffusers/quantizers/quantization_config.py +669 -0
diffusers/quantizers/torchao/__init__.py +15 -0
diffusers/quantizers/torchao/torchao_quantizer.py +292 -0
diffusers/schedulers/__init__.py +12 -2
diffusers/schedulers/deprecated/__init__.py +1 -1
diffusers/schedulers/deprecated/scheduling_karras_ve.py +25 -25
diffusers/schedulers/scheduling_amused.py +5 -5
diffusers/schedulers/scheduling_consistency_decoder.py +11 -11
diffusers/schedulers/scheduling_consistency_models.py +23 -25
diffusers/schedulers/scheduling_cosine_dpmsolver_multistep.py +572 -0
diffusers/schedulers/scheduling_ddim.py +27 -26
diffusers/schedulers/scheduling_ddim_cogvideox.py +452 -0
diffusers/schedulers/scheduling_ddim_flax.py +2 -1
diffusers/schedulers/scheduling_ddim_inverse.py +16 -16
diffusers/schedulers/scheduling_ddim_parallel.py +32 -31
diffusers/schedulers/scheduling_ddpm.py +27 -30
diffusers/schedulers/scheduling_ddpm_flax.py +7 -3
diffusers/schedulers/scheduling_ddpm_parallel.py +33 -36
diffusers/schedulers/scheduling_ddpm_wuerstchen.py +14 -14
diffusers/schedulers/scheduling_deis_multistep.py +150 -50
diffusers/schedulers/scheduling_dpm_cogvideox.py +489 -0
diffusers/schedulers/scheduling_dpmsolver_multistep.py +221 -84
diffusers/schedulers/scheduling_dpmsolver_multistep_flax.py +2 -2
diffusers/schedulers/scheduling_dpmsolver_multistep_inverse.py +158 -52
diffusers/schedulers/scheduling_dpmsolver_sde.py +153 -34
diffusers/schedulers/scheduling_dpmsolver_singlestep.py +275 -86
diffusers/schedulers/scheduling_edm_dpmsolver_multistep.py +81 -57
diffusers/schedulers/scheduling_edm_euler.py +62 -39
diffusers/schedulers/scheduling_euler_ancestral_discrete.py +30 -29
diffusers/schedulers/scheduling_euler_discrete.py +255 -74
diffusers/schedulers/scheduling_flow_match_euler_discrete.py +458 -0
diffusers/schedulers/scheduling_flow_match_heun_discrete.py +320 -0
diffusers/schedulers/scheduling_heun_discrete.py +174 -46
diffusers/schedulers/scheduling_ipndm.py +9 -9
diffusers/schedulers/scheduling_k_dpm_2_ancestral_discrete.py +138 -29
diffusers/schedulers/scheduling_k_dpm_2_discrete.py +132 -26
diffusers/schedulers/scheduling_karras_ve_flax.py +6 -6
diffusers/schedulers/scheduling_lcm.py +23 -29
diffusers/schedulers/scheduling_lms_discrete.py +105 -28
diffusers/schedulers/scheduling_pndm.py +20 -20
diffusers/schedulers/scheduling_repaint.py +21 -21
diffusers/schedulers/scheduling_sasolver.py +157 -60
diffusers/schedulers/scheduling_sde_ve.py +19 -19
diffusers/schedulers/scheduling_tcd.py +41 -36
diffusers/schedulers/scheduling_unclip.py +19 -16
diffusers/schedulers/scheduling_unipc_multistep.py +243 -47
diffusers/schedulers/scheduling_utils.py +12 -5
diffusers/schedulers/scheduling_utils_flax.py +1 -3
diffusers/schedulers/scheduling_vq_diffusion.py +10 -10
diffusers/training_utils.py +214 -30
diffusers/utils/__init__.py +17 -1
diffusers/utils/constants.py +3 -0
diffusers/utils/doc_utils.py +1 -0
diffusers/utils/dummy_pt_objects.py +592 -7
diffusers/utils/dummy_torch_and_torchsde_objects.py +15 -0
diffusers/utils/dummy_torch_and_transformers_and_sentencepiece_objects.py +47 -0
diffusers/utils/dummy_torch_and_transformers_objects.py +1001 -71
diffusers/utils/dynamic_modules_utils.py +34 -29
diffusers/utils/export_utils.py +50 -6
diffusers/utils/hub_utils.py +131 -17
diffusers/utils/import_utils.py +210 -8
diffusers/utils/loading_utils.py +118 -5
diffusers/utils/logging.py +4 -2
diffusers/utils/peft_utils.py +37 -7
diffusers/utils/state_dict_utils.py +13 -2
diffusers/utils/testing_utils.py +193 -11
diffusers/utils/torch_utils.py +4 -0
diffusers/video_processor.py +113 -0
{diffusers-0.27.1.dist-info → diffusers-0.32.2.dist-info}/METADATA +82 -91
diffusers-0.32.2.dist-info/RECORD +550 -0
{diffusers-0.27.1.dist-info → diffusers-0.32.2.dist-info}/WHEEL +1 -1
diffusers/loaders/autoencoder.py +0 -146
diffusers/loaders/controlnet.py +0 -136
diffusers/loaders/lora.py +0 -1349
diffusers/models/prior_transformer.py +0 -12
diffusers/models/t5_film_transformer.py +0 -70
diffusers/models/transformer_2d.py +0 -25
diffusers/models/transformer_temporal.py +0 -34
diffusers/models/unet_1d.py +0 -26
diffusers/models/unet_1d_blocks.py +0 -203
diffusers/models/unet_2d.py +0 -27
diffusers/models/unet_2d_blocks.py +0 -375
diffusers/models/unet_2d_condition.py +0 -25
diffusers-0.27.1.dist-info/RECORD +0 -399
{diffusers-0.27.1.dist-info → diffusers-0.32.2.dist-info}/LICENSE +0 -0
{diffusers-0.27.1.dist-info → diffusers-0.32.2.dist-info}/entry_points.txt +0 -0
{diffusers-0.27.1.dist-info → diffusers-0.32.2.dist-info}/top_level.txt +0 -0

diffusers/loaders/lora_conversion_utils.py CHANGED Viewed

@@ -14,7 +14,9 @@
 import re
-from ..utils import logging
+import torch
+from ..utils import is_peft_version, logging
 logger = logging.get_logger(__name__)
@@ -123,153 +125,100 @@ def _maybe_map_sgm_blocks_to_diffusers(state_dict, unet_config, delimiter="_", b
     return new_state_dict
-def _convert_kohya_lora_to_diffusers(state_dict, unet_name="unet", text_encoder_name="text_encoder"):
+def _convert_non_diffusers_lora_to_diffusers(state_dict, unet_name="unet", text_encoder_name="text_encoder"):
+    """
+    Converts a non-Diffusers LoRA state dict to a Diffusers compatible state dict.
+    Args:
+        state_dict (`dict`): The state dict to convert.
+        unet_name (`str`, optional): The name of the U-Net module in the Diffusers model. Defaults to "unet".
+        text_encoder_name (`str`, optional): The name of the text encoder module in the Diffusers model. Defaults to
+            "text_encoder".
+    Returns:
+        `tuple`: A tuple containing the converted state dict and a dictionary of alphas.
+    """
     unet_state_dict = {}
     te_state_dict = {}
     te2_state_dict = {}
     network_alphas = {}
-    # every down weight has a corresponding up weight and potentially an alpha weight
-    lora_keys = [k for k in state_dict.keys() if k.endswith("lora_down.weight")]
-    for key in lora_keys:
+    # Check for DoRA-enabled LoRAs.
+    dora_present_in_unet = any("dora_scale" in k and "lora_unet_" in k for k in state_dict)
+    dora_present_in_te = any("dora_scale" in k and ("lora_te_" in k or "lora_te1_" in k) for k in state_dict)
+    dora_present_in_te2 = any("dora_scale" in k and "lora_te2_" in k for k in state_dict)
+    if dora_present_in_unet or dora_present_in_te or dora_present_in_te2:
+        if is_peft_version("<", "0.9.0"):
+            raise ValueError(
+                "You need `peft` 0.9.0 at least to use DoRA-enabled LoRAs. Please upgrade your installation of `peft`."
+            )
+    # Iterate over all LoRA weights.
+    all_lora_keys = list(state_dict.keys())
+    for key in all_lora_keys:
+        if not key.endswith("lora_down.weight"):
+            continue
+        # Extract LoRA name.
         lora_name = key.split(".")[0]
+        # Find corresponding up weight and alpha.
         lora_name_up = lora_name + ".lora_up.weight"
         lora_name_alpha = lora_name + ".alpha"
+        # Handle U-Net LoRAs.
         if lora_name.startswith("lora_unet_"):
-            diffusers_name = key.replace("lora_unet_", "").replace("_", ".")
+            diffusers_name = _convert_unet_lora_key(key)
-            if "input.blocks" in diffusers_name:
-                diffusers_name = diffusers_name.replace("input.blocks", "down_blocks")
-            else:
-                diffusers_name = diffusers_name.replace("down.blocks", "down_blocks")
+            # Store down and up weights.
+            unet_state_dict[diffusers_name] = state_dict.pop(key)
+            unet_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-            if "middle.block" in diffusers_name:
-                diffusers_name = diffusers_name.replace("middle.block", "mid_block")
-            else:
-                diffusers_name = diffusers_name.replace("mid.block", "mid_block")
-            if "output.blocks" in diffusers_name:
-                diffusers_name = diffusers_name.replace("output.blocks", "up_blocks")
-            else:
-                diffusers_name = diffusers_name.replace("up.blocks", "up_blocks")
-            diffusers_name = diffusers_name.replace("transformer.blocks", "transformer_blocks")
-            diffusers_name = diffusers_name.replace("to.q.lora", "to_q_lora")
-            diffusers_name = diffusers_name.replace("to.k.lora", "to_k_lora")
-            diffusers_name = diffusers_name.replace("to.v.lora", "to_v_lora")
-            diffusers_name = diffusers_name.replace("to.out.0.lora", "to_out_lora")
-            diffusers_name = diffusers_name.replace("proj.in", "proj_in")
-            diffusers_name = diffusers_name.replace("proj.out", "proj_out")
-            diffusers_name = diffusers_name.replace("emb.layers", "time_emb_proj")
-            # SDXL specificity.
-            if "emb" in diffusers_name and "time.emb.proj" not in diffusers_name:
-                pattern = r"\.\d+(?=\D*$)"
-                diffusers_name = re.sub(pattern, "", diffusers_name, count=1)
-            if ".in." in diffusers_name:
-                diffusers_name = diffusers_name.replace("in.layers.2", "conv1")
-            if ".out." in diffusers_name:
-                diffusers_name = diffusers_name.replace("out.layers.3", "conv2")
-            if "downsamplers" in diffusers_name or "upsamplers" in diffusers_name:
-                diffusers_name = diffusers_name.replace("op", "conv")
-            if "skip" in diffusers_name:
-                diffusers_name = diffusers_name.replace("skip.connection", "conv_shortcut")
-            # LyCORIS specificity.
-            if "time.emb.proj" in diffusers_name:
-                diffusers_name = diffusers_name.replace("time.emb.proj", "time_emb_proj")
-            if "conv.shortcut" in diffusers_name:
-                diffusers_name = diffusers_name.replace("conv.shortcut", "conv_shortcut")
-            # General coverage.
-            if "transformer_blocks" in diffusers_name:
-                if "attn1" in diffusers_name or "attn2" in diffusers_name:
-                    diffusers_name = diffusers_name.replace("attn1", "attn1.processor")
-                    diffusers_name = diffusers_name.replace("attn2", "attn2.processor")
-                    unet_state_dict[diffusers_name] = state_dict.pop(key)
-                    unet_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-                elif "ff" in diffusers_name:
-                    unet_state_dict[diffusers_name] = state_dict.pop(key)
-                    unet_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-            elif any(key in diffusers_name for key in ("proj_in", "proj_out")):
-                unet_state_dict[diffusers_name] = state_dict.pop(key)
-                unet_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-            else:
-                unet_state_dict[diffusers_name] = state_dict.pop(key)
-                unet_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-        elif lora_name.startswith("lora_te_"):
-            diffusers_name = key.replace("lora_te_", "").replace("_", ".")
-            diffusers_name = diffusers_name.replace("text.model", "text_model")
-            diffusers_name = diffusers_name.replace("self.attn", "self_attn")
-            diffusers_name = diffusers_name.replace("q.proj.lora", "to_q_lora")
-            diffusers_name = diffusers_name.replace("k.proj.lora", "to_k_lora")
-            diffusers_name = diffusers_name.replace("v.proj.lora", "to_v_lora")
-            diffusers_name = diffusers_name.replace("out.proj.lora", "to_out_lora")
-            if "self_attn" in diffusers_name:
-                te_state_dict[diffusers_name] = state_dict.pop(key)
-                te_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-            elif "mlp" in diffusers_name:
-                # Be aware that this is the new diffusers convention and the rest of the code might
-                # not utilize it yet.
-                diffusers_name = diffusers_name.replace(".lora.", ".lora_linear_layer.")
-                te_state_dict[diffusers_name] = state_dict.pop(key)
-                te_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
+            # Store DoRA scale if present.
+            if dora_present_in_unet:
+                dora_scale_key_to_replace = "_lora.down." if "_lora.down." in diffusers_name else ".lora.down."
+                unet_state_dict[
+                    diffusers_name.replace(dora_scale_key_to_replace, ".lora_magnitude_vector.")
+                ] = state_dict.pop(key.replace("lora_down.weight", "dora_scale"))
-        # (sayakpaul): Duplicate code. Needs to be cleaned.
-        elif lora_name.startswith("lora_te1_"):
-            diffusers_name = key.replace("lora_te1_", "").replace("_", ".")
-            diffusers_name = diffusers_name.replace("text.model", "text_model")
-            diffusers_name = diffusers_name.replace("self.attn", "self_attn")
-            diffusers_name = diffusers_name.replace("q.proj.lora", "to_q_lora")
-            diffusers_name = diffusers_name.replace("k.proj.lora", "to_k_lora")
-            diffusers_name = diffusers_name.replace("v.proj.lora", "to_v_lora")
-            diffusers_name = diffusers_name.replace("out.proj.lora", "to_out_lora")
-            if "self_attn" in diffusers_name:
-                te_state_dict[diffusers_name] = state_dict.pop(key)
-                te_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-            elif "mlp" in diffusers_name:
-                # Be aware that this is the new diffusers convention and the rest of the code might
-                # not utilize it yet.
-                diffusers_name = diffusers_name.replace(".lora.", ".lora_linear_layer.")
+        # Handle text encoder LoRAs.
+        elif lora_name.startswith(("lora_te_", "lora_te1_", "lora_te2_")):
+            diffusers_name = _convert_text_encoder_lora_key(key, lora_name)
+            # Store down and up weights for te or te2.
+            if lora_name.startswith(("lora_te_", "lora_te1_")):
                 te_state_dict[diffusers_name] = state_dict.pop(key)
                 te_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-        # (sayakpaul): Duplicate code. Needs to be cleaned.
-        elif lora_name.startswith("lora_te2_"):
-            diffusers_name = key.replace("lora_te2_", "").replace("_", ".")
-            diffusers_name = diffusers_name.replace("text.model", "text_model")
-            diffusers_name = diffusers_name.replace("self.attn", "self_attn")
-            diffusers_name = diffusers_name.replace("q.proj.lora", "to_q_lora")
-            diffusers_name = diffusers_name.replace("k.proj.lora", "to_k_lora")
-            diffusers_name = diffusers_name.replace("v.proj.lora", "to_v_lora")
-            diffusers_name = diffusers_name.replace("out.proj.lora", "to_out_lora")
-            if "self_attn" in diffusers_name:
-                te2_state_dict[diffusers_name] = state_dict.pop(key)
-                te2_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-            elif "mlp" in diffusers_name:
-                # Be aware that this is the new diffusers convention and the rest of the code might
-                # not utilize it yet.
-                diffusers_name = diffusers_name.replace(".lora.", ".lora_linear_layer.")
+            else:
                 te2_state_dict[diffusers_name] = state_dict.pop(key)
                 te2_state_dict[diffusers_name.replace(".down.", ".up.")] = state_dict.pop(lora_name_up)
-        # Rename the alphas so that they can be mapped appropriately.
+            # Store DoRA scale if present.
+            if dora_present_in_te or dora_present_in_te2:
+                dora_scale_key_to_replace_te = (
+                    "_lora.down." if "_lora.down." in diffusers_name else ".lora_linear_layer."
+                )
+                if lora_name.startswith(("lora_te_", "lora_te1_")):
+                    te_state_dict[
+                        diffusers_name.replace(dora_scale_key_to_replace_te, ".lora_magnitude_vector.")
+                    ] = state_dict.pop(key.replace("lora_down.weight", "dora_scale"))
+                elif lora_name.startswith("lora_te2_"):
+                    te2_state_dict[
+                        diffusers_name.replace(dora_scale_key_to_replace_te, ".lora_magnitude_vector.")
+                    ] = state_dict.pop(key.replace("lora_down.weight", "dora_scale"))
+        # Store alpha if present.
         if lora_name_alpha in state_dict:
             alpha = state_dict.pop(lora_name_alpha).item()
-            if lora_name_alpha.startswith("lora_unet_"):
-                prefix = "unet."
-            elif lora_name_alpha.startswith(("lora_te_", "lora_te1_")):
-                prefix = "text_encoder."
-            else:
-                prefix = "text_encoder_2."
-            new_name = prefix + diffusers_name.split(".lora.")[0] + ".alpha"
-            network_alphas.update({new_name: alpha})
+            network_alphas.update(_get_alpha_name(lora_name_alpha, diffusers_name, alpha))
+    # Check if any keys remain.
     if len(state_dict) > 0:
-        raise ValueError(f"The following keys have not been correctly be renamed: \n\n {', '.join(state_dict.keys())}")
+        raise ValueError(f"The following keys have not been correctly renamed: \n\n {', '.join(state_dict.keys())}")
+    logger.info("Non-diffusers checkpoint detected.")
-    logger.info("Kohya-style checkpoint detected.")
+    # Construct final state dict.
     unet_state_dict = {f"{unet_name}.{module_name}": params for module_name, params in unet_state_dict.items()}
     te_state_dict = {f"{text_encoder_name}.{module_name}": params for module_name, params in te_state_dict.items()}
     te2_state_dict = (
@@ -282,3 +231,920 @@ def _convert_kohya_lora_to_diffusers(state_dict, unet_name="unet", text_encoder_
     new_state_dict = {**unet_state_dict, **te_state_dict}
     return new_state_dict, network_alphas
+def _convert_unet_lora_key(key):
+    """
+    Converts a U-Net LoRA key to a Diffusers compatible key.
+    """
+    diffusers_name = key.replace("lora_unet_", "").replace("_", ".")
+    # Replace common U-Net naming patterns.
+    diffusers_name = diffusers_name.replace("input.blocks", "down_blocks")
+    diffusers_name = diffusers_name.replace("down.blocks", "down_blocks")
+    diffusers_name = diffusers_name.replace("middle.block", "mid_block")
+    diffusers_name = diffusers_name.replace("mid.block", "mid_block")
+    diffusers_name = diffusers_name.replace("output.blocks", "up_blocks")
+    diffusers_name = diffusers_name.replace("up.blocks", "up_blocks")
+    diffusers_name = diffusers_name.replace("transformer.blocks", "transformer_blocks")
+    diffusers_name = diffusers_name.replace("to.q.lora", "to_q_lora")
+    diffusers_name = diffusers_name.replace("to.k.lora", "to_k_lora")
+    diffusers_name = diffusers_name.replace("to.v.lora", "to_v_lora")
+    diffusers_name = diffusers_name.replace("to.out.0.lora", "to_out_lora")
+    diffusers_name = diffusers_name.replace("proj.in", "proj_in")
+    diffusers_name = diffusers_name.replace("proj.out", "proj_out")
+    diffusers_name = diffusers_name.replace("emb.layers", "time_emb_proj")
+    # SDXL specific conversions.
+    if "emb" in diffusers_name and "time.emb.proj" not in diffusers_name:
+        pattern = r"\.\d+(?=\D*$)"
+        diffusers_name = re.sub(pattern, "", diffusers_name, count=1)
+    if ".in." in diffusers_name:
+        diffusers_name = diffusers_name.replace("in.layers.2", "conv1")
+    if ".out." in diffusers_name:
+        diffusers_name = diffusers_name.replace("out.layers.3", "conv2")
+    if "downsamplers" in diffusers_name or "upsamplers" in diffusers_name:
+        diffusers_name = diffusers_name.replace("op", "conv")
+    if "skip" in diffusers_name:
+        diffusers_name = diffusers_name.replace("skip.connection", "conv_shortcut")
+    # LyCORIS specific conversions.
+    if "time.emb.proj" in diffusers_name:
+        diffusers_name = diffusers_name.replace("time.emb.proj", "time_emb_proj")
+    if "conv.shortcut" in diffusers_name:
+        diffusers_name = diffusers_name.replace("conv.shortcut", "conv_shortcut")
+    # General conversions.
+    if "transformer_blocks" in diffusers_name:
+        if "attn1" in diffusers_name or "attn2" in diffusers_name:
+            diffusers_name = diffusers_name.replace("attn1", "attn1.processor")
+            diffusers_name = diffusers_name.replace("attn2", "attn2.processor")
+        elif "ff" in diffusers_name:
+            pass
+    elif any(key in diffusers_name for key in ("proj_in", "proj_out")):
+        pass
+    else:
+        pass
+    return diffusers_name
+def _convert_text_encoder_lora_key(key, lora_name):
+    """
+    Converts a text encoder LoRA key to a Diffusers compatible key.
+    """
+    if lora_name.startswith(("lora_te_", "lora_te1_")):
+        key_to_replace = "lora_te_" if lora_name.startswith("lora_te_") else "lora_te1_"
+    else:
+        key_to_replace = "lora_te2_"
+    diffusers_name = key.replace(key_to_replace, "").replace("_", ".")
+    diffusers_name = diffusers_name.replace("text.model", "text_model")
+    diffusers_name = diffusers_name.replace("self.attn", "self_attn")
+    diffusers_name = diffusers_name.replace("q.proj.lora", "to_q_lora")
+    diffusers_name = diffusers_name.replace("k.proj.lora", "to_k_lora")
+    diffusers_name = diffusers_name.replace("v.proj.lora", "to_v_lora")
+    diffusers_name = diffusers_name.replace("out.proj.lora", "to_out_lora")
+    diffusers_name = diffusers_name.replace("text.projection", "text_projection")
+    if "self_attn" in diffusers_name or "text_projection" in diffusers_name:
+        pass
+    elif "mlp" in diffusers_name:
+        # Be aware that this is the new diffusers convention and the rest of the code might
+        # not utilize it yet.
+        diffusers_name = diffusers_name.replace(".lora.", ".lora_linear_layer.")
+    return diffusers_name
+def _get_alpha_name(lora_name_alpha, diffusers_name, alpha):
+    """
+    Gets the correct alpha name for the Diffusers model.
+    """
+    if lora_name_alpha.startswith("lora_unet_"):
+        prefix = "unet."
+    elif lora_name_alpha.startswith(("lora_te_", "lora_te1_")):
+        prefix = "text_encoder."
+    else:
+        prefix = "text_encoder_2."
+    new_name = prefix + diffusers_name.split(".lora.")[0] + ".alpha"
+    return {new_name: alpha}
+# The utilities under `_convert_kohya_flux_lora_to_diffusers()`
+# are taken from https://github.com/kohya-ss/sd-scripts/blob/a61cf73a5cb5209c3f4d1a3688dd276a4dfd1ecb/networks/convert_flux_lora.py
+# All credits go to `kohya-ss`.
+def _convert_kohya_flux_lora_to_diffusers(state_dict):
+    def _convert_to_ai_toolkit(sds_sd, ait_sd, sds_key, ait_key):
+        if sds_key + ".lora_down.weight" not in sds_sd:
+            return
+        down_weight = sds_sd.pop(sds_key + ".lora_down.weight")
+        # scale weight by alpha and dim
+        rank = down_weight.shape[0]
+        alpha = sds_sd.pop(sds_key + ".alpha").item()  # alpha is scalar
+        scale = alpha / rank  # LoRA is scaled by 'alpha / rank' in forward pass, so we need to scale it back here
+        # calculate scale_down and scale_up to keep the same value. if scale is 4, scale_down is 2 and scale_up is 2
+        scale_down = scale
+        scale_up = 1.0
+        while scale_down * 2 < scale_up:
+            scale_down *= 2
+            scale_up /= 2
+        ait_sd[ait_key + ".lora_A.weight"] = down_weight * scale_down
+        ait_sd[ait_key + ".lora_B.weight"] = sds_sd.pop(sds_key + ".lora_up.weight") * scale_up
+    def _convert_to_ai_toolkit_cat(sds_sd, ait_sd, sds_key, ait_keys, dims=None):
+        if sds_key + ".lora_down.weight" not in sds_sd:
+            return
+        down_weight = sds_sd.pop(sds_key + ".lora_down.weight")
+        up_weight = sds_sd.pop(sds_key + ".lora_up.weight")
+        sd_lora_rank = down_weight.shape[0]
+        # scale weight by alpha and dim
+        alpha = sds_sd.pop(sds_key + ".alpha")
+        scale = alpha / sd_lora_rank
+        # calculate scale_down and scale_up
+        scale_down = scale
+        scale_up = 1.0
+        while scale_down * 2 < scale_up:
+            scale_down *= 2
+            scale_up /= 2
+        down_weight = down_weight * scale_down
+        up_weight = up_weight * scale_up
+        # calculate dims if not provided
+        num_splits = len(ait_keys)
+        if dims is None:
+            dims = [up_weight.shape[0] // num_splits] * num_splits
+        else:
+            assert sum(dims) == up_weight.shape[0]
+        # check upweight is sparse or not
+        is_sparse = False
+        if sd_lora_rank % num_splits == 0:
+            ait_rank = sd_lora_rank // num_splits
+            is_sparse = True
+            i = 0
+            for j in range(len(dims)):
+                for k in range(len(dims)):
+                    if j == k:
+                        continue
+                    is_sparse = is_sparse and torch.all(
+                        up_weight[i : i + dims[j], k * ait_rank : (k + 1) * ait_rank] == 0
+                    )
+                i += dims[j]
+            if is_sparse:
+                logger.info(f"weight is sparse: {sds_key}")
+        # make ai-toolkit weight
+        ait_down_keys = [k + ".lora_A.weight" for k in ait_keys]
+        ait_up_keys = [k + ".lora_B.weight" for k in ait_keys]
+        if not is_sparse:
+            # down_weight is copied to each split
+            ait_sd.update({k: down_weight for k in ait_down_keys})
+            # up_weight is split to each split
+            ait_sd.update({k: v for k, v in zip(ait_up_keys, torch.split(up_weight, dims, dim=0))})  # noqa: C416
+        else:
+            # down_weight is chunked to each split
+            ait_sd.update({k: v for k, v in zip(ait_down_keys, torch.chunk(down_weight, num_splits, dim=0))})  # noqa: C416
+            # up_weight is sparse: only non-zero values are copied to each split
+            i = 0
+            for j in range(len(dims)):
+                ait_sd[ait_up_keys[j]] = up_weight[i : i + dims[j], j * ait_rank : (j + 1) * ait_rank].contiguous()
+                i += dims[j]
+    def _convert_sd_scripts_to_ai_toolkit(sds_sd):
+        ait_sd = {}
+        for i in range(19):
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_img_attn_proj",
+                f"transformer.transformer_blocks.{i}.attn.to_out.0",
+            )
+            _convert_to_ai_toolkit_cat(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_img_attn_qkv",
+                [
+                    f"transformer.transformer_blocks.{i}.attn.to_q",
+                    f"transformer.transformer_blocks.{i}.attn.to_k",
+                    f"transformer.transformer_blocks.{i}.attn.to_v",
+                ],
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_img_mlp_0",
+                f"transformer.transformer_blocks.{i}.ff.net.0.proj",
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_img_mlp_2",
+                f"transformer.transformer_blocks.{i}.ff.net.2",
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_img_mod_lin",
+                f"transformer.transformer_blocks.{i}.norm1.linear",
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_txt_attn_proj",
+                f"transformer.transformer_blocks.{i}.attn.to_add_out",
+            )
+            _convert_to_ai_toolkit_cat(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_txt_attn_qkv",
+                [
+                    f"transformer.transformer_blocks.{i}.attn.add_q_proj",
+                    f"transformer.transformer_blocks.{i}.attn.add_k_proj",
+                    f"transformer.transformer_blocks.{i}.attn.add_v_proj",
+                ],
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_txt_mlp_0",
+                f"transformer.transformer_blocks.{i}.ff_context.net.0.proj",
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_txt_mlp_2",
+                f"transformer.transformer_blocks.{i}.ff_context.net.2",
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_double_blocks_{i}_txt_mod_lin",
+                f"transformer.transformer_blocks.{i}.norm1_context.linear",
+            )
+        for i in range(38):
+            _convert_to_ai_toolkit_cat(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_single_blocks_{i}_linear1",
+                [
+                    f"transformer.single_transformer_blocks.{i}.attn.to_q",
+                    f"transformer.single_transformer_blocks.{i}.attn.to_k",
+                    f"transformer.single_transformer_blocks.{i}.attn.to_v",
+                    f"transformer.single_transformer_blocks.{i}.proj_mlp",
+                ],
+                dims=[3072, 3072, 3072, 12288],
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_single_blocks_{i}_linear2",
+                f"transformer.single_transformer_blocks.{i}.proj_out",
+            )
+            _convert_to_ai_toolkit(
+                sds_sd,
+                ait_sd,
+                f"lora_unet_single_blocks_{i}_modulation_lin",
+                f"transformer.single_transformer_blocks.{i}.norm.linear",
+            )
+        remaining_keys = list(sds_sd.keys())
+        te_state_dict = {}
+        if remaining_keys:
+            if not all(k.startswith("lora_te1") for k in remaining_keys):
+                raise ValueError(f"Incompatible keys detected: \n\n {', '.join(remaining_keys)}")
+            for key in remaining_keys:
+                if not key.endswith("lora_down.weight"):
+                    continue
+                lora_name = key.split(".")[0]
+                lora_name_up = f"{lora_name}.lora_up.weight"
+                lora_name_alpha = f"{lora_name}.alpha"
+                diffusers_name = _convert_text_encoder_lora_key(key, lora_name)
+                if lora_name.startswith(("lora_te_", "lora_te1_")):
+                    down_weight = sds_sd.pop(key)
+                    sd_lora_rank = down_weight.shape[0]
+                    te_state_dict[diffusers_name] = down_weight
+                    te_state_dict[diffusers_name.replace(".down.", ".up.")] = sds_sd.pop(lora_name_up)
+                if lora_name_alpha in sds_sd:
+                    alpha = sds_sd.pop(lora_name_alpha).item()
+                    scale = alpha / sd_lora_rank
+                    scale_down = scale
+                    scale_up = 1.0
+                    while scale_down * 2 < scale_up:
+                        scale_down *= 2
+                        scale_up /= 2
+                    te_state_dict[diffusers_name] *= scale_down
+                    te_state_dict[diffusers_name.replace(".down.", ".up.")] *= scale_up
+        if len(sds_sd) > 0:
+            logger.warning(f"Unsupported keys for ai-toolkit: {sds_sd.keys()}")
+        if te_state_dict:
+            te_state_dict = {f"text_encoder.{module_name}": params for module_name, params in te_state_dict.items()}
+        new_state_dict = {**ait_sd, **te_state_dict}
+        return new_state_dict
+    return _convert_sd_scripts_to_ai_toolkit(state_dict)
+# Adapted from https://gist.github.com/Leommm-byte/6b331a1e9bd53271210b26543a7065d6
+# Some utilities were reused from
+# https://github.com/kohya-ss/sd-scripts/blob/a61cf73a5cb5209c3f4d1a3688dd276a4dfd1ecb/networks/convert_flux_lora.py
+def _convert_xlabs_flux_lora_to_diffusers(old_state_dict):
+    new_state_dict = {}
+    orig_keys = list(old_state_dict.keys())
+    def handle_qkv(sds_sd, ait_sd, sds_key, ait_keys, dims=None):
+        down_weight = sds_sd.pop(sds_key)
+        up_weight = sds_sd.pop(sds_key.replace(".down.weight", ".up.weight"))
+        # calculate dims if not provided
+        num_splits = len(ait_keys)
+        if dims is None:
+            dims = [up_weight.shape[0] // num_splits] * num_splits
+        else:
+            assert sum(dims) == up_weight.shape[0]
+        # make ai-toolkit weight
+        ait_down_keys = [k + ".lora_A.weight" for k in ait_keys]
+        ait_up_keys = [k + ".lora_B.weight" for k in ait_keys]
+        # down_weight is copied to each split
+        ait_sd.update({k: down_weight for k in ait_down_keys})
+        # up_weight is split to each split
+        ait_sd.update({k: v for k, v in zip(ait_up_keys, torch.split(up_weight, dims, dim=0))})  # noqa: C416
+    for old_key in orig_keys:
+        # Handle double_blocks
+        if old_key.startswith(("diffusion_model.double_blocks", "double_blocks")):
+            block_num = re.search(r"double_blocks\.(\d+)", old_key).group(1)
+            new_key = f"transformer.transformer_blocks.{block_num}"
+            if "processor.proj_lora1" in old_key:
+                new_key += ".attn.to_out.0"
+            elif "processor.proj_lora2" in old_key:
+                new_key += ".attn.to_add_out"
+            # Handle text latents.
+            elif "processor.qkv_lora2" in old_key and "up" not in old_key:
+                handle_qkv(
+                    old_state_dict,
+                    new_state_dict,
+                    old_key,
+                    [
+                        f"transformer.transformer_blocks.{block_num}.attn.add_q_proj",
+                        f"transformer.transformer_blocks.{block_num}.attn.add_k_proj",
+                        f"transformer.transformer_blocks.{block_num}.attn.add_v_proj",
+                    ],
+                )
+                # continue
+            # Handle image latents.
+            elif "processor.qkv_lora1" in old_key and "up" not in old_key:
+                handle_qkv(
+                    old_state_dict,
+                    new_state_dict,
+                    old_key,
+                    [
+                        f"transformer.transformer_blocks.{block_num}.attn.to_q",
+                        f"transformer.transformer_blocks.{block_num}.attn.to_k",
+                        f"transformer.transformer_blocks.{block_num}.attn.to_v",
+                    ],
+                )
+                # continue
+            if "down" in old_key:
+                new_key += ".lora_A.weight"
+            elif "up" in old_key:
+                new_key += ".lora_B.weight"
+        # Handle single_blocks
+        elif old_key.startswith(("diffusion_model.single_blocks", "single_blocks")):
+            block_num = re.search(r"single_blocks\.(\d+)", old_key).group(1)
+            new_key = f"transformer.single_transformer_blocks.{block_num}"
+            if "proj_lora" in old_key:
+                new_key += ".proj_out"
+            elif "qkv_lora" in old_key and "up" not in old_key:
+                handle_qkv(
+                    old_state_dict,
+                    new_state_dict,
+                    old_key,
+                    [
+                        f"transformer.single_transformer_blocks.{block_num}.attn.to_q",
+                        f"transformer.single_transformer_blocks.{block_num}.attn.to_k",
+                        f"transformer.single_transformer_blocks.{block_num}.attn.to_v",
+                    ],
+                )
+            if "down" in old_key:
+                new_key += ".lora_A.weight"
+            elif "up" in old_key:
+                new_key += ".lora_B.weight"
+        else:
+            # Handle other potential key patterns here
+            new_key = old_key
+        # Since we already handle qkv above.
+        if "qkv" not in old_key:
+            new_state_dict[new_key] = old_state_dict.pop(old_key)
+    if len(old_state_dict) > 0:
+        raise ValueError(f"`old_state_dict` should be at this point but has: {list(old_state_dict.keys())}.")
+    return new_state_dict
+def _convert_bfl_flux_control_lora_to_diffusers(original_state_dict):
+    converted_state_dict = {}
+    original_state_dict_keys = list(original_state_dict.keys())
+    num_layers = 19
+    num_single_layers = 38
+    inner_dim = 3072
+    mlp_ratio = 4.0
+    def swap_scale_shift(weight):
+        shift, scale = weight.chunk(2, dim=0)
+        new_weight = torch.cat([scale, shift], dim=0)
+        return new_weight
+    for lora_key in ["lora_A", "lora_B"]:
+        ## time_text_embed.timestep_embedder <-  time_in
+        converted_state_dict[
+            f"time_text_embed.timestep_embedder.linear_1.{lora_key}.weight"
+        ] = original_state_dict.pop(f"time_in.in_layer.{lora_key}.weight")
+        if f"time_in.in_layer.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[
+                f"time_text_embed.timestep_embedder.linear_1.{lora_key}.bias"
+            ] = original_state_dict.pop(f"time_in.in_layer.{lora_key}.bias")
+        converted_state_dict[
+            f"time_text_embed.timestep_embedder.linear_2.{lora_key}.weight"
+        ] = original_state_dict.pop(f"time_in.out_layer.{lora_key}.weight")
+        if f"time_in.out_layer.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[
+                f"time_text_embed.timestep_embedder.linear_2.{lora_key}.bias"
+            ] = original_state_dict.pop(f"time_in.out_layer.{lora_key}.bias")
+        ## time_text_embed.text_embedder <- vector_in
+        converted_state_dict[f"time_text_embed.text_embedder.linear_1.{lora_key}.weight"] = original_state_dict.pop(
+            f"vector_in.in_layer.{lora_key}.weight"
+        )
+        if f"vector_in.in_layer.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[f"time_text_embed.text_embedder.linear_1.{lora_key}.bias"] = original_state_dict.pop(
+                f"vector_in.in_layer.{lora_key}.bias"
+            )
+        converted_state_dict[f"time_text_embed.text_embedder.linear_2.{lora_key}.weight"] = original_state_dict.pop(
+            f"vector_in.out_layer.{lora_key}.weight"
+        )
+        if f"vector_in.out_layer.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[f"time_text_embed.text_embedder.linear_2.{lora_key}.bias"] = original_state_dict.pop(
+                f"vector_in.out_layer.{lora_key}.bias"
+            )
+        # guidance
+        has_guidance = any("guidance" in k for k in original_state_dict)
+        if has_guidance:
+            converted_state_dict[
+                f"time_text_embed.guidance_embedder.linear_1.{lora_key}.weight"
+            ] = original_state_dict.pop(f"guidance_in.in_layer.{lora_key}.weight")
+            if f"guidance_in.in_layer.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[
+                    f"time_text_embed.guidance_embedder.linear_1.{lora_key}.bias"
+                ] = original_state_dict.pop(f"guidance_in.in_layer.{lora_key}.bias")
+            converted_state_dict[
+                f"time_text_embed.guidance_embedder.linear_2.{lora_key}.weight"
+            ] = original_state_dict.pop(f"guidance_in.out_layer.{lora_key}.weight")
+            if f"guidance_in.out_layer.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[
+                    f"time_text_embed.guidance_embedder.linear_2.{lora_key}.bias"
+                ] = original_state_dict.pop(f"guidance_in.out_layer.{lora_key}.bias")
+        # context_embedder
+        converted_state_dict[f"context_embedder.{lora_key}.weight"] = original_state_dict.pop(
+            f"txt_in.{lora_key}.weight"
+        )
+        if f"txt_in.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[f"context_embedder.{lora_key}.bias"] = original_state_dict.pop(
+                f"txt_in.{lora_key}.bias"
+            )
+        # x_embedder
+        converted_state_dict[f"x_embedder.{lora_key}.weight"] = original_state_dict.pop(f"img_in.{lora_key}.weight")
+        if f"img_in.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[f"x_embedder.{lora_key}.bias"] = original_state_dict.pop(f"img_in.{lora_key}.bias")
+    # double transformer blocks
+    for i in range(num_layers):
+        block_prefix = f"transformer_blocks.{i}."
+        for lora_key in ["lora_A", "lora_B"]:
+            # norms
+            converted_state_dict[f"{block_prefix}norm1.linear.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.img_mod.lin.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.img_mod.lin.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}norm1.linear.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.img_mod.lin.{lora_key}.bias"
+                )
+            converted_state_dict[f"{block_prefix}norm1_context.linear.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.txt_mod.lin.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.txt_mod.lin.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}norm1_context.linear.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.txt_mod.lin.{lora_key}.bias"
+                )
+            # Q, K, V
+            if lora_key == "lora_A":
+                sample_lora_weight = original_state_dict.pop(f"double_blocks.{i}.img_attn.qkv.{lora_key}.weight")
+                converted_state_dict[f"{block_prefix}attn.to_v.{lora_key}.weight"] = torch.cat([sample_lora_weight])
+                converted_state_dict[f"{block_prefix}attn.to_q.{lora_key}.weight"] = torch.cat([sample_lora_weight])
+                converted_state_dict[f"{block_prefix}attn.to_k.{lora_key}.weight"] = torch.cat([sample_lora_weight])
+                context_lora_weight = original_state_dict.pop(f"double_blocks.{i}.txt_attn.qkv.{lora_key}.weight")
+                converted_state_dict[f"{block_prefix}attn.add_q_proj.{lora_key}.weight"] = torch.cat(
+                    [context_lora_weight]
+                )
+                converted_state_dict[f"{block_prefix}attn.add_k_proj.{lora_key}.weight"] = torch.cat(
+                    [context_lora_weight]
+                )
+                converted_state_dict[f"{block_prefix}attn.add_v_proj.{lora_key}.weight"] = torch.cat(
+                    [context_lora_weight]
+                )
+            else:
+                sample_q, sample_k, sample_v = torch.chunk(
+                    original_state_dict.pop(f"double_blocks.{i}.img_attn.qkv.{lora_key}.weight"), 3, dim=0
+                )
+                converted_state_dict[f"{block_prefix}attn.to_q.{lora_key}.weight"] = torch.cat([sample_q])
+                converted_state_dict[f"{block_prefix}attn.to_k.{lora_key}.weight"] = torch.cat([sample_k])
+                converted_state_dict[f"{block_prefix}attn.to_v.{lora_key}.weight"] = torch.cat([sample_v])
+                context_q, context_k, context_v = torch.chunk(
+                    original_state_dict.pop(f"double_blocks.{i}.txt_attn.qkv.{lora_key}.weight"), 3, dim=0
+                )
+                converted_state_dict[f"{block_prefix}attn.add_q_proj.{lora_key}.weight"] = torch.cat([context_q])
+                converted_state_dict[f"{block_prefix}attn.add_k_proj.{lora_key}.weight"] = torch.cat([context_k])
+                converted_state_dict[f"{block_prefix}attn.add_v_proj.{lora_key}.weight"] = torch.cat([context_v])
+            if f"double_blocks.{i}.img_attn.qkv.{lora_key}.bias" in original_state_dict_keys:
+                sample_q_bias, sample_k_bias, sample_v_bias = torch.chunk(
+                    original_state_dict.pop(f"double_blocks.{i}.img_attn.qkv.{lora_key}.bias"), 3, dim=0
+                )
+                converted_state_dict[f"{block_prefix}attn.to_q.{lora_key}.bias"] = torch.cat([sample_q_bias])
+                converted_state_dict[f"{block_prefix}attn.to_k.{lora_key}.bias"] = torch.cat([sample_k_bias])
+                converted_state_dict[f"{block_prefix}attn.to_v.{lora_key}.bias"] = torch.cat([sample_v_bias])
+            if f"double_blocks.{i}.txt_attn.qkv.{lora_key}.bias" in original_state_dict_keys:
+                context_q_bias, context_k_bias, context_v_bias = torch.chunk(
+                    original_state_dict.pop(f"double_blocks.{i}.txt_attn.qkv.{lora_key}.bias"), 3, dim=0
+                )
+                converted_state_dict[f"{block_prefix}attn.add_q_proj.{lora_key}.bias"] = torch.cat([context_q_bias])
+                converted_state_dict[f"{block_prefix}attn.add_k_proj.{lora_key}.bias"] = torch.cat([context_k_bias])
+                converted_state_dict[f"{block_prefix}attn.add_v_proj.{lora_key}.bias"] = torch.cat([context_v_bias])
+            # ff img_mlp
+            converted_state_dict[f"{block_prefix}ff.net.0.proj.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.img_mlp.0.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.img_mlp.0.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}ff.net.0.proj.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.img_mlp.0.{lora_key}.bias"
+                )
+            converted_state_dict[f"{block_prefix}ff.net.2.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.img_mlp.2.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.img_mlp.2.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}ff.net.2.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.img_mlp.2.{lora_key}.bias"
+                )
+            converted_state_dict[f"{block_prefix}ff_context.net.0.proj.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.txt_mlp.0.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.txt_mlp.0.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}ff_context.net.0.proj.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.txt_mlp.0.{lora_key}.bias"
+                )
+            converted_state_dict[f"{block_prefix}ff_context.net.2.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.txt_mlp.2.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.txt_mlp.2.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}ff_context.net.2.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.txt_mlp.2.{lora_key}.bias"
+                )
+            # output projections.
+            converted_state_dict[f"{block_prefix}attn.to_out.0.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.img_attn.proj.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.img_attn.proj.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}attn.to_out.0.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.img_attn.proj.{lora_key}.bias"
+                )
+            converted_state_dict[f"{block_prefix}attn.to_add_out.{lora_key}.weight"] = original_state_dict.pop(
+                f"double_blocks.{i}.txt_attn.proj.{lora_key}.weight"
+            )
+            if f"double_blocks.{i}.txt_attn.proj.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}attn.to_add_out.{lora_key}.bias"] = original_state_dict.pop(
+                    f"double_blocks.{i}.txt_attn.proj.{lora_key}.bias"
+                )
+        # qk_norm
+        converted_state_dict[f"{block_prefix}attn.norm_q.weight"] = original_state_dict.pop(
+            f"double_blocks.{i}.img_attn.norm.query_norm.scale"
+        )
+        converted_state_dict[f"{block_prefix}attn.norm_k.weight"] = original_state_dict.pop(
+            f"double_blocks.{i}.img_attn.norm.key_norm.scale"
+        )
+        converted_state_dict[f"{block_prefix}attn.norm_added_q.weight"] = original_state_dict.pop(
+            f"double_blocks.{i}.txt_attn.norm.query_norm.scale"
+        )
+        converted_state_dict[f"{block_prefix}attn.norm_added_k.weight"] = original_state_dict.pop(
+            f"double_blocks.{i}.txt_attn.norm.key_norm.scale"
+        )
+    # single transfomer blocks
+    for i in range(num_single_layers):
+        block_prefix = f"single_transformer_blocks.{i}."
+        for lora_key in ["lora_A", "lora_B"]:
+            # norm.linear  <- single_blocks.0.modulation.lin
+            converted_state_dict[f"{block_prefix}norm.linear.{lora_key}.weight"] = original_state_dict.pop(
+                f"single_blocks.{i}.modulation.lin.{lora_key}.weight"
+            )
+            if f"single_blocks.{i}.modulation.lin.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}norm.linear.{lora_key}.bias"] = original_state_dict.pop(
+                    f"single_blocks.{i}.modulation.lin.{lora_key}.bias"
+                )
+            # Q, K, V, mlp
+            mlp_hidden_dim = int(inner_dim * mlp_ratio)
+            split_size = (inner_dim, inner_dim, inner_dim, mlp_hidden_dim)
+            if lora_key == "lora_A":
+                lora_weight = original_state_dict.pop(f"single_blocks.{i}.linear1.{lora_key}.weight")
+                converted_state_dict[f"{block_prefix}attn.to_q.{lora_key}.weight"] = torch.cat([lora_weight])
+                converted_state_dict[f"{block_prefix}attn.to_k.{lora_key}.weight"] = torch.cat([lora_weight])
+                converted_state_dict[f"{block_prefix}attn.to_v.{lora_key}.weight"] = torch.cat([lora_weight])
+                converted_state_dict[f"{block_prefix}proj_mlp.{lora_key}.weight"] = torch.cat([lora_weight])
+                if f"single_blocks.{i}.linear1.{lora_key}.bias" in original_state_dict_keys:
+                    lora_bias = original_state_dict.pop(f"single_blocks.{i}.linear1.{lora_key}.bias")
+                    converted_state_dict[f"{block_prefix}attn.to_q.{lora_key}.bias"] = torch.cat([lora_bias])
+                    converted_state_dict[f"{block_prefix}attn.to_k.{lora_key}.bias"] = torch.cat([lora_bias])
+                    converted_state_dict[f"{block_prefix}attn.to_v.{lora_key}.bias"] = torch.cat([lora_bias])
+                    converted_state_dict[f"{block_prefix}proj_mlp.{lora_key}.bias"] = torch.cat([lora_bias])
+            else:
+                q, k, v, mlp = torch.split(
+                    original_state_dict.pop(f"single_blocks.{i}.linear1.{lora_key}.weight"), split_size, dim=0
+                )
+                converted_state_dict[f"{block_prefix}attn.to_q.{lora_key}.weight"] = torch.cat([q])
+                converted_state_dict[f"{block_prefix}attn.to_k.{lora_key}.weight"] = torch.cat([k])
+                converted_state_dict[f"{block_prefix}attn.to_v.{lora_key}.weight"] = torch.cat([v])
+                converted_state_dict[f"{block_prefix}proj_mlp.{lora_key}.weight"] = torch.cat([mlp])
+                if f"single_blocks.{i}.linear1.{lora_key}.bias" in original_state_dict_keys:
+                    q_bias, k_bias, v_bias, mlp_bias = torch.split(
+                        original_state_dict.pop(f"single_blocks.{i}.linear1.{lora_key}.bias"), split_size, dim=0
+                    )
+                    converted_state_dict[f"{block_prefix}attn.to_q.{lora_key}.bias"] = torch.cat([q_bias])
+                    converted_state_dict[f"{block_prefix}attn.to_k.{lora_key}.bias"] = torch.cat([k_bias])
+                    converted_state_dict[f"{block_prefix}attn.to_v.{lora_key}.bias"] = torch.cat([v_bias])
+                    converted_state_dict[f"{block_prefix}proj_mlp.{lora_key}.bias"] = torch.cat([mlp_bias])
+            # output projections.
+            converted_state_dict[f"{block_prefix}proj_out.{lora_key}.weight"] = original_state_dict.pop(
+                f"single_blocks.{i}.linear2.{lora_key}.weight"
+            )
+            if f"single_blocks.{i}.linear2.{lora_key}.bias" in original_state_dict_keys:
+                converted_state_dict[f"{block_prefix}proj_out.{lora_key}.bias"] = original_state_dict.pop(
+                    f"single_blocks.{i}.linear2.{lora_key}.bias"
+                )
+        # qk norm
+        converted_state_dict[f"{block_prefix}attn.norm_q.weight"] = original_state_dict.pop(
+            f"single_blocks.{i}.norm.query_norm.scale"
+        )
+        converted_state_dict[f"{block_prefix}attn.norm_k.weight"] = original_state_dict.pop(
+            f"single_blocks.{i}.norm.key_norm.scale"
+        )
+    for lora_key in ["lora_A", "lora_B"]:
+        converted_state_dict[f"proj_out.{lora_key}.weight"] = original_state_dict.pop(
+            f"final_layer.linear.{lora_key}.weight"
+        )
+        if f"final_layer.linear.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[f"proj_out.{lora_key}.bias"] = original_state_dict.pop(
+                f"final_layer.linear.{lora_key}.bias"
+            )
+        converted_state_dict[f"norm_out.linear.{lora_key}.weight"] = swap_scale_shift(
+            original_state_dict.pop(f"final_layer.adaLN_modulation.1.{lora_key}.weight")
+        )
+        if f"final_layer.adaLN_modulation.1.{lora_key}.bias" in original_state_dict_keys:
+            converted_state_dict[f"norm_out.linear.{lora_key}.bias"] = swap_scale_shift(
+                original_state_dict.pop(f"final_layer.adaLN_modulation.1.{lora_key}.bias")
+            )
+    if len(original_state_dict) > 0:
+        raise ValueError(f"`original_state_dict` should be empty at this point but has {original_state_dict.keys()=}.")
+    for key in list(converted_state_dict.keys()):
+        converted_state_dict[f"transformer.{key}"] = converted_state_dict.pop(key)
+    return converted_state_dict
+def _convert_hunyuan_video_lora_to_diffusers(original_state_dict):
+    converted_state_dict = {k: original_state_dict.pop(k) for k in list(original_state_dict.keys())}
+    def remap_norm_scale_shift_(key, state_dict):
+        weight = state_dict.pop(key)
+        shift, scale = weight.chunk(2, dim=0)
+        new_weight = torch.cat([scale, shift], dim=0)
+        state_dict[key.replace("final_layer.adaLN_modulation.1", "norm_out.linear")] = new_weight
+    def remap_txt_in_(key, state_dict):
+        def rename_key(key):
+            new_key = key.replace("individual_token_refiner.blocks", "token_refiner.refiner_blocks")
+            new_key = new_key.replace("adaLN_modulation.1", "norm_out.linear")
+            new_key = new_key.replace("txt_in", "context_embedder")
+            new_key = new_key.replace("t_embedder.mlp.0", "time_text_embed.timestep_embedder.linear_1")
+            new_key = new_key.replace("t_embedder.mlp.2", "time_text_embed.timestep_embedder.linear_2")
+            new_key = new_key.replace("c_embedder", "time_text_embed.text_embedder")
+            new_key = new_key.replace("mlp", "ff")
+            return new_key
+        if "self_attn_qkv" in key:
+            weight = state_dict.pop(key)
+            to_q, to_k, to_v = weight.chunk(3, dim=0)
+            state_dict[rename_key(key.replace("self_attn_qkv", "attn.to_q"))] = to_q
+            state_dict[rename_key(key.replace("self_attn_qkv", "attn.to_k"))] = to_k
+            state_dict[rename_key(key.replace("self_attn_qkv", "attn.to_v"))] = to_v
+        else:
+            state_dict[rename_key(key)] = state_dict.pop(key)
+    def remap_img_attn_qkv_(key, state_dict):
+        weight = state_dict.pop(key)
+        if "lora_A" in key:
+            state_dict[key.replace("img_attn_qkv", "attn.to_q")] = weight
+            state_dict[key.replace("img_attn_qkv", "attn.to_k")] = weight
+            state_dict[key.replace("img_attn_qkv", "attn.to_v")] = weight
+        else:
+            to_q, to_k, to_v = weight.chunk(3, dim=0)
+            state_dict[key.replace("img_attn_qkv", "attn.to_q")] = to_q
+            state_dict[key.replace("img_attn_qkv", "attn.to_k")] = to_k
+            state_dict[key.replace("img_attn_qkv", "attn.to_v")] = to_v
+    def remap_txt_attn_qkv_(key, state_dict):
+        weight = state_dict.pop(key)
+        if "lora_A" in key:
+            state_dict[key.replace("txt_attn_qkv", "attn.add_q_proj")] = weight
+            state_dict[key.replace("txt_attn_qkv", "attn.add_k_proj")] = weight
+            state_dict[key.replace("txt_attn_qkv", "attn.add_v_proj")] = weight
+        else:
+            to_q, to_k, to_v = weight.chunk(3, dim=0)
+            state_dict[key.replace("txt_attn_qkv", "attn.add_q_proj")] = to_q
+            state_dict[key.replace("txt_attn_qkv", "attn.add_k_proj")] = to_k
+            state_dict[key.replace("txt_attn_qkv", "attn.add_v_proj")] = to_v
+    def remap_single_transformer_blocks_(key, state_dict):
+        hidden_size = 3072
+        if "linear1.lora_A.weight" in key or "linear1.lora_B.weight" in key:
+            linear1_weight = state_dict.pop(key)
+            if "lora_A" in key:
+                new_key = key.replace("single_blocks", "single_transformer_blocks").removesuffix(
+                    ".linear1.lora_A.weight"
+                )
+                state_dict[f"{new_key}.attn.to_q.lora_A.weight"] = linear1_weight
+                state_dict[f"{new_key}.attn.to_k.lora_A.weight"] = linear1_weight
+                state_dict[f"{new_key}.attn.to_v.lora_A.weight"] = linear1_weight
+                state_dict[f"{new_key}.proj_mlp.lora_A.weight"] = linear1_weight
+            else:
+                split_size = (hidden_size, hidden_size, hidden_size, linear1_weight.size(0) - 3 * hidden_size)
+                q, k, v, mlp = torch.split(linear1_weight, split_size, dim=0)
+                new_key = key.replace("single_blocks", "single_transformer_blocks").removesuffix(
+                    ".linear1.lora_B.weight"
+                )
+                state_dict[f"{new_key}.attn.to_q.lora_B.weight"] = q
+                state_dict[f"{new_key}.attn.to_k.lora_B.weight"] = k
+                state_dict[f"{new_key}.attn.to_v.lora_B.weight"] = v
+                state_dict[f"{new_key}.proj_mlp.lora_B.weight"] = mlp
+        elif "linear1.lora_A.bias" in key or "linear1.lora_B.bias" in key:
+            linear1_bias = state_dict.pop(key)
+            if "lora_A" in key:
+                new_key = key.replace("single_blocks", "single_transformer_blocks").removesuffix(
+                    ".linear1.lora_A.bias"
+                )
+                state_dict[f"{new_key}.attn.to_q.lora_A.bias"] = linear1_bias
+                state_dict[f"{new_key}.attn.to_k.lora_A.bias"] = linear1_bias
+                state_dict[f"{new_key}.attn.to_v.lora_A.bias"] = linear1_bias
+                state_dict[f"{new_key}.proj_mlp.lora_A.bias"] = linear1_bias
+            else:
+                split_size = (hidden_size, hidden_size, hidden_size, linear1_bias.size(0) - 3 * hidden_size)
+                q_bias, k_bias, v_bias, mlp_bias = torch.split(linear1_bias, split_size, dim=0)
+                new_key = key.replace("single_blocks", "single_transformer_blocks").removesuffix(
+                    ".linear1.lora_B.bias"
+                )
+                state_dict[f"{new_key}.attn.to_q.lora_B.bias"] = q_bias
+                state_dict[f"{new_key}.attn.to_k.lora_B.bias"] = k_bias
+                state_dict[f"{new_key}.attn.to_v.lora_B.bias"] = v_bias
+                state_dict[f"{new_key}.proj_mlp.lora_B.bias"] = mlp_bias
+        else:
+            new_key = key.replace("single_blocks", "single_transformer_blocks")
+            new_key = new_key.replace("linear2", "proj_out")
+            new_key = new_key.replace("q_norm", "attn.norm_q")
+            new_key = new_key.replace("k_norm", "attn.norm_k")
+            state_dict[new_key] = state_dict.pop(key)
+    TRANSFORMER_KEYS_RENAME_DICT = {
+        "img_in": "x_embedder",
+        "time_in.mlp.0": "time_text_embed.timestep_embedder.linear_1",
+        "time_in.mlp.2": "time_text_embed.timestep_embedder.linear_2",
+        "guidance_in.mlp.0": "time_text_embed.guidance_embedder.linear_1",
+        "guidance_in.mlp.2": "time_text_embed.guidance_embedder.linear_2",
+        "vector_in.in_layer": "time_text_embed.text_embedder.linear_1",
+        "vector_in.out_layer": "time_text_embed.text_embedder.linear_2",
+        "double_blocks": "transformer_blocks",
+        "img_attn_q_norm": "attn.norm_q",
+        "img_attn_k_norm": "attn.norm_k",
+        "img_attn_proj": "attn.to_out.0",
+        "txt_attn_q_norm": "attn.norm_added_q",
+        "txt_attn_k_norm": "attn.norm_added_k",
+        "txt_attn_proj": "attn.to_add_out",
+        "img_mod.linear": "norm1.linear",
+        "img_norm1": "norm1.norm",
+        "img_norm2": "norm2",
+        "img_mlp": "ff",
+        "txt_mod.linear": "norm1_context.linear",
+        "txt_norm1": "norm1.norm",
+        "txt_norm2": "norm2_context",
+        "txt_mlp": "ff_context",
+        "self_attn_proj": "attn.to_out.0",
+        "modulation.linear": "norm.linear",
+        "pre_norm": "norm.norm",
+        "final_layer.norm_final": "norm_out.norm",
+        "final_layer.linear": "proj_out",
+        "fc1": "net.0.proj",
+        "fc2": "net.2",
+        "input_embedder": "proj_in",
+    }
+    TRANSFORMER_SPECIAL_KEYS_REMAP = {
+        "txt_in": remap_txt_in_,
+        "img_attn_qkv": remap_img_attn_qkv_,
+        "txt_attn_qkv": remap_txt_attn_qkv_,
+        "single_blocks": remap_single_transformer_blocks_,
+        "final_layer.adaLN_modulation.1": remap_norm_scale_shift_,
+    }
+    # Some folks attempt to make their state dict compatible with diffusers by adding "transformer." prefix to all keys
+    # and use their custom code. To make sure both "original" and "attempted diffusers" loras work as expected, we make
+    # sure that both follow the same initial format by stripping off the "transformer." prefix.
+    for key in list(converted_state_dict.keys()):
+        if key.startswith("transformer."):
+            converted_state_dict[key[len("transformer.") :]] = converted_state_dict.pop(key)
+        if key.startswith("diffusion_model."):
+            converted_state_dict[key[len("diffusion_model.") :]] = converted_state_dict.pop(key)
+    # Rename and remap the state dict keys
+    for key in list(converted_state_dict.keys()):
+        new_key = key[:]
+        for replace_key, rename_key in TRANSFORMER_KEYS_RENAME_DICT.items():
+            new_key = new_key.replace(replace_key, rename_key)
+        converted_state_dict[new_key] = converted_state_dict.pop(key)
+    for key in list(converted_state_dict.keys()):
+        for special_key, handler_fn_inplace in TRANSFORMER_SPECIAL_KEYS_REMAP.items():
+            if special_key not in key:
+                continue
+            handler_fn_inplace(key, converted_state_dict)
+    # Add back the "transformer." prefix
+    for key in list(converted_state_dict.keys()):
+        converted_state_dict[f"transformer.{key}"] = converted_state_dict.pop(key)
+    return converted_state_dict

diffusers 0.27.1__py3-none-any.whl → 0.32.2__py3-none-any.whl

diffusers 0.27.1py3-none-any.whl → 0.32.2py3-none-any.whl