diffsynth 2.0.18__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {diffsynth-2.0.18 → diffsynth-2.1.0}/PKG-INFO +6 -1
- {diffsynth-2.0.18 → diffsynth-2.1.0}/README.md +145 -1
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/configs/model_configs.py +113 -3
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/configs/vram_management_module_maps.py +49 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/__init__.py +1 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/device/npu_compatible_device.py +2 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/loader/config.py +59 -1
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/loader/file.py +5 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/loader/model.py +64 -8
- diffsynth-2.1.0/diffsynth/core/quant/__init__.py +3 -0
- diffsynth-2.1.0/diffsynth/core/quant/backends/__init__.py +1 -0
- diffsynth-2.1.0/diffsynth/core/quant/backends/bitsandbytes.py +152 -0
- diffsynth-2.1.0/diffsynth/core/quant/backends/torchao.py +121 -0
- diffsynth-2.1.0/diffsynth/core/quant/base.py +127 -0
- diffsynth-2.1.0/diffsynth/core/quant/config.py +425 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/vram/layers.py +156 -30
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/base_pipeline.py +5 -4
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/flow_match.py +10 -1
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/loss.py +31 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ideogram4_dit.py +56 -38
- diffsynth-2.1.0/diffsynth/models/ideogram4_text_encoder.py +137 -0
- diffsynth-2.1.0/diffsynth/models/lingbot_video_dit.py +623 -0
- diffsynth-2.1.0/diffsynth/models/minimax_constant.py +23 -0
- diffsynth-2.1.0/diffsynth/models/minimax_h3_audio_vae.py +481 -0
- diffsynth-2.1.0/diffsynth/models/minimax_h3_dit.py +397 -0
- diffsynth-2.1.0/diffsynth/models/minimax_h3_text_encoder.py +231 -0
- diffsynth-2.1.0/diffsynth/models/minimax_h3_video_vae.py +552 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/model_loader.py +19 -4
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/qwen_image_vae.py +33 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_dit.py +1 -1
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/flux_image.py +6 -0
- diffsynth-2.1.0/diffsynth/pipelines/lingbot_video.py +426 -0
- diffsynth-2.1.0/diffsynth/pipelines/minimax_h3_audio_video.py +1062 -0
- diffsynth-2.1.0/diffsynth/utils/data/minimax_h3.py +105 -0
- diffsynth-2.1.0/diffsynth/utils/lora/flux_timestep.py +150 -0
- diffsynth-2.1.0/diffsynth/utils/state_dict_converters/ideogram4_text_encoder.py +3 -0
- diffsynth-2.1.0/diffsynth/utils/state_dict_converters/lingbot_video_dit.py +11 -0
- diffsynth-2.1.0/diffsynth/utils/state_dict_converters/minimax_h3_audio_vae.py +9 -0
- diffsynth-2.1.0/diffsynth/utils/state_dict_converters/minimax_h3_text_encoder.py +27 -0
- diffsynth-2.1.0/diffsynth/utils/state_dict_converters/minimax_h3_video_vae.py +11 -0
- diffsynth-2.1.0/diffsynth/utils/tile/__init__.py +1 -0
- diffsynth-2.1.0/diffsynth/utils/tile/tile_worker.py +55 -0
- diffsynth-2.1.0/diffsynth/utils/xfuser/__init__.py +1 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth.egg-info/PKG-INFO +6 -1
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth.egg-info/SOURCES.txt +23 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth.egg-info/requires.txt +6 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/pyproject.toml +8 -2
- diffsynth-2.0.18/diffsynth/models/ideogram4_text_encoder.py +0 -353
- diffsynth-2.0.18/diffsynth/utils/xfuser/__init__.py +0 -1
- {diffsynth-2.0.18 → diffsynth-2.1.0}/LICENSE +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/configs/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/attention/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/attention/attention.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/data/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/data/operators.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/data/unified_dataset.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/device/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/gradient/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/gradient/gradient_checkpoint.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/loader/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/npu_patch/npu_fused_operator.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/offload_training/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/offload_training/manager.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/offload_training/memory_buffer.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/offload_training/offloader.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/vram/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/vram/disk_map.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/core/vram/initialization.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/ddim_scheduler.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/dmd2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/logger.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/parsers.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/runner.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/template.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/diffusion/training_module.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/aesthetic.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/base.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/bioclip.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/clip.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/fid.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/hpsv2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/hpsv3.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/image_reward.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/lpips.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/pickscore.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/qwen_image_bench.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/unified_reward_2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/metrics/unified_reward_edit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ace_step_conditioner.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ace_step_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ace_step_residual_fsq.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ace_step_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ace_step_tokenizer.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ace_step_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/aesthetic.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/anima_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/bioclip.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/boogu_image_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/clip.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/demucs.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/dinov3_image_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ernie_image_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ernie_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/fid.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux2_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux2_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux2_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_controlnet.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_infiniteyou.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_ipadapter.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_lora_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_lora_patcher.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_text_encoder_clip.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_text_encoder_t5.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/flux_value_control.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/general_modules.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/hidream_common.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/hidream_o1_image_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/hpsv2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/hpsv3.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ideogram4_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/image_reward.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/joyai_image_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/joyai_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/krea2_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/krea2_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/longcat_video_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/lpips.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ltx2_audio_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ltx2_common.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ltx2_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ltx2_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ltx2_upsampler.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/ltx2_video_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/mova_audio_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/mova_audio_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/mova_dual_tower_bridge.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/nexus_gen.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/nexus_gen_ar_model.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/nexus_gen_projector.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/pickscore.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/qwen_image_bench.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/qwen_image_controlnet.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/qwen_image_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/qwen_image_image2lora.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/qwen_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/sd_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/siglip2_image_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/stable_diffusion_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/stable_diffusion_unet.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/stable_diffusion_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/stable_diffusion_xl_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/stable_diffusion_xl_unet.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/step1x_connector.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/step1x_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/unified_reward_2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/unified_reward_edit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_animate_adapter.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_camera_controller.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_dit_s2v.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_image_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_mot.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_motion_controller.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_vace.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wan_video_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wantodance.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/wav2vec.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/z_image_controlnet.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/z_image_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/z_image_image2lora.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/models/z_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/ace_step.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/anima_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/boogu_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/ernie_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/flux2_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/hidream_o1_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/ideogram4.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/joyai_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/krea2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/ltx2_audio_video.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/mova_audio_video.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/qwen_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/stable_diffusion.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/stable_diffusion_xl.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/wan_video.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/pipelines/z_image.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/controlnet/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/controlnet/annotator.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/controlnet/controlnet_input.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/data/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/data/audio.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/data/audio_video.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/data/media_io_ltx2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/demucs/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/dequantizer/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/lora/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/lora/flux.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/lora/general.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/lora/krea2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/lora/merge.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/lora/reset_rank.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/lora/sdxl.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/ses/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/ses/ses.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/__init__.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ace_step_conditioner.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ace_step_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ace_step_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ace_step_tokenizer.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/anima_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/dino_v3.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ernie_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux2_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux_controlnet.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux_infiniteyou.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux_ipadapter.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux_text_encoder_clip.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux_text_encoder_t5.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/flux_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/image_metrics.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/joyai_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/krea2_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/krea2_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ltx2_audio_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ltx2_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ltx2_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/ltx2_video_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/nexus_gen.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/nexus_gen_projector.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/qwen_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/sdxl.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/sdxl_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/sdxl_text_encoder_2.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/sdxl_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/stable_diffusion_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/stable_diffusion_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/stable_diffusion_xl_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/step1x_connector.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/wan_video_animate_adapter.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/wan_video_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/wan_video_image_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/wan_video_mot.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/wan_video_vace.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/wan_video_vae.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/wans2v_audio_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/z_image_dit.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/state_dict_converters/z_image_text_encoder.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/utils/xfuser/xdit_context_parallel.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth/version.py +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth.egg-info/dependency_links.txt +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/diffsynth.egg-info/top_level.txt +0 -0
- {diffsynth-2.0.18 → diffsynth-2.1.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: diffsynth
|
|
3
|
-
Version: 2.0
|
|
3
|
+
Version: 2.1.0
|
|
4
4
|
Summary: Enjoy the magic of Diffusion models!
|
|
5
5
|
Author: ModelScope Team
|
|
6
6
|
License: Apache-2.0
|
|
@@ -37,10 +37,15 @@ Requires-Dist: av; extra == "audio"
|
|
|
37
37
|
Requires-Dist: torchaudio; extra == "audio"
|
|
38
38
|
Requires-Dist: torchcodec; extra == "audio"
|
|
39
39
|
Requires-Dist: librosa; extra == "audio"
|
|
40
|
+
Provides-Extra: quant
|
|
41
|
+
Requires-Dist: bitsandbytes; extra == "quant"
|
|
42
|
+
Requires-Dist: torchao; extra == "quant"
|
|
40
43
|
Provides-Extra: all
|
|
41
44
|
Requires-Dist: av; extra == "all"
|
|
42
45
|
Requires-Dist: torchaudio; extra == "all"
|
|
43
46
|
Requires-Dist: torchcodec; extra == "all"
|
|
44
47
|
Requires-Dist: librosa; extra == "all"
|
|
45
48
|
Requires-Dist: streamlit; extra == "all"
|
|
49
|
+
Requires-Dist: bitsandbytes; extra == "all"
|
|
50
|
+
Requires-Dist: torchao; extra == "all"
|
|
46
51
|
Dynamic: license-file
|
|
@@ -36,6 +36,10 @@ We believe that a well-developed open-source code framework can lower the thresh
|
|
|
36
36
|
|
|
37
37
|
> Currently, the development personnel of this project are limited, with most of the work handled by [Artiprocher](https://github.com/Artiprocher) and [mi804](https://github.com/mi804). Therefore, the progress of new feature development will be relatively slow, and the speed of responding to and resolving issues is limited. We apologize for this and ask developers to understand.
|
|
38
38
|
|
|
39
|
+
- **August 3, 2026** MiniMax-H3 open-sourced, welcome a new member to the video model family! Support includes text-to-video-audio generation, keyframe-guided generation, reference-driven generation, low VRAM inference, and NF4-quantized inference. For details, please refer to the [documentation](/docs/en/Model_Details/MiniMax-H3.md) and [example code](/examples/minimax_h3/).
|
|
40
|
+
|
|
41
|
+
- **July 28, 2026** LingBot-Video open-sourced, welcome a new member to the video model family! This release includes two variants, Dense-1.3B and MoE-30B-A3B (30B total parameters, ~3B active per token), both supporting text-to-video, image-to-video and text-to-image generation, low VRAM inference, and LoRA / full training capabilities. For details, please refer to the [documentation](/docs/en/Model_Details/LingBot-Video.md) and [example code](/examples/lingbot_video/). Huge thanks to [NancyFyong](https://github.com/NancyFyong) for contributing the integration of this model!
|
|
42
|
+
|
|
39
43
|
- **July 21, 2026** We have open-sourced [DiffSynth-Studio Model Integration Skills](https://www.modelscope.cn/collections/DiffSynth-Studio/DiffSynth-Studio-Model-Integration-Skills). This is a composable collection of Agent Skills that automates the entire workflow of integrating external diffusion models into DiffSynth-Studio, significantly improving the standardization and efficiency of model integration. Get started with the [example](https://www.modelscope.cn/skills/DiffSynth-Studio/diffsynth-integrator/file/view/master/example.md?status=1)!
|
|
40
44
|
|
|
41
45
|
- **June 29, 2026** Boogu-Image open-sourced. Support includes text-to-image generation, image editing, low VRAM inference, and training capabilities. For details, please refer to the [documentation](/docs/en/Model_Details/Boogu-Image.md) and [example code](/examples/boogu_image/).
|
|
@@ -267,7 +271,7 @@ DiffSynth-Studio redesigns the inference and training pipelines for mainstream D
|
|
|
267
271
|
>
|
|
268
272
|
> ```python
|
|
269
273
|
> import os
|
|
270
|
-
> os.environ["
|
|
274
|
+
> os.environ["MODELSCOPE_ENDPOINT"] = "https://modelscope.ai"
|
|
271
275
|
> ```
|
|
272
276
|
>
|
|
273
277
|
> To download models from other sources, please modify the environment variable [DIFFSYNTH_DOWNLOAD_SOURCE](/docs/en/Pipeline_Usage/Environment_Variables.md#diffsynth_download_source).
|
|
@@ -1447,6 +1451,146 @@ Example code for Wan is available at: [/examples/wanvideo/](/examples/wanvideo/)
|
|
|
1447
1451
|
|
|
1448
1452
|
</details>
|
|
1449
1453
|
|
|
1454
|
+
#### LingBot-Video: [/docs/en/Model_Details/LingBot-Video.md](/docs/en/Model_Details/LingBot-Video.md)
|
|
1455
|
+
|
|
1456
|
+
<details>
|
|
1457
|
+
|
|
1458
|
+
<summary>Quick Start</summary>
|
|
1459
|
+
|
|
1460
|
+
Running the following code will quickly load the [Robbyant/lingbot-video-dense-1.3b](https://modelscope.cn/models/Robbyant/lingbot-video-dense-1.3b) model and perform inference. VRAM management is enabled, and the framework will automatically control the loading of model parameters based on available VRAM. The model can run with a minimum of 6GB VRAM.
|
|
1461
|
+
|
|
1462
|
+
```python
|
|
1463
|
+
import torch
|
|
1464
|
+
import json
|
|
1465
|
+
from diffsynth.utils.data import save_video, VideoData
|
|
1466
|
+
from diffsynth.pipelines.lingbot_video import LingBotVideoPipeline, ModelConfig
|
|
1467
|
+
from modelscope import dataset_snapshot_download
|
|
1468
|
+
|
|
1469
|
+
vram_config = {
|
|
1470
|
+
"offload_dtype": "disk",
|
|
1471
|
+
"offload_device": "disk",
|
|
1472
|
+
"onload_dtype": torch.float8_e4m3fn,
|
|
1473
|
+
"onload_device": "cpu",
|
|
1474
|
+
"preparing_dtype": torch.float8_e4m3fn,
|
|
1475
|
+
"preparing_device": "cuda",
|
|
1476
|
+
"computation_dtype": torch.bfloat16,
|
|
1477
|
+
"computation_device": "cuda",
|
|
1478
|
+
}
|
|
1479
|
+
|
|
1480
|
+
pipe = LingBotVideoPipeline.from_pretrained(
|
|
1481
|
+
torch_dtype=torch.bfloat16,
|
|
1482
|
+
device="cuda",
|
|
1483
|
+
model_configs=[
|
|
1484
|
+
ModelConfig(model_id="Robbyant/lingbot-video-dense-1.3b", origin_file_pattern="transformer/diffusion_pytorch_model.safetensors", **vram_config),
|
|
1485
|
+
ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern="*.safetensors", **vram_config),
|
|
1486
|
+
ModelConfig(model_id="Robbyant/lingbot-video-dense-1.3b", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config),
|
|
1487
|
+
],
|
|
1488
|
+
processor_config=ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern=""),
|
|
1489
|
+
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 0.5,
|
|
1490
|
+
)
|
|
1491
|
+
|
|
1492
|
+
dataset_snapshot_download(
|
|
1493
|
+
dataset_id="DiffSynth-Studio/diffsynth_example_dataset",
|
|
1494
|
+
local_dir="data/diffsynth_example_dataset",
|
|
1495
|
+
allow_file_pattern="lingbot_video/lingbot-video-dense-1.3b_t2v/*",
|
|
1496
|
+
)
|
|
1497
|
+
with open("data/diffsynth_example_dataset/lingbot_video/lingbot-video-dense-1.3b_t2v/t2v_example_1.json", "r", encoding="utf-8") as f:
|
|
1498
|
+
caption = json.load(f)
|
|
1499
|
+
|
|
1500
|
+
video = pipe(
|
|
1501
|
+
prompt=caption,
|
|
1502
|
+
negative_prompt=pipe.default_negative_prompt,
|
|
1503
|
+
height=480, width=832, num_frames=81,
|
|
1504
|
+
num_inference_steps=40, cfg_scale=3.0,
|
|
1505
|
+
seed=0,
|
|
1506
|
+
)
|
|
1507
|
+
save_video(video, "video.mp4", fps=15, quality=10)
|
|
1508
|
+
```
|
|
1509
|
+
|
|
1510
|
+
</details>
|
|
1511
|
+
|
|
1512
|
+
<details>
|
|
1513
|
+
|
|
1514
|
+
<summary>Examples</summary>
|
|
1515
|
+
|
|
1516
|
+
Example code for LingBot-Video is available at: [/examples/lingbot_video/](/examples/lingbot_video/)
|
|
1517
|
+
|
|
1518
|
+
| Model ID | Inference | Low VRAM Inference | Full Training | Full Training Validation | LoRA Training | LoRA Training Validation |
|
|
1519
|
+
|-|-|-|-|-|-|-|
|
|
1520
|
+
|[Robbyant/lingbot-video-dense-1.3b: T2V](https://modelscope.cn/models/Robbyant/lingbot-video-dense-1.3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-dense-1.3b_t2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-dense-1.3b_t2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-dense-1.3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-dense-1.3b_t2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-dense-1.3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-dense-1.3b_t2v.py)|
|
|
1521
|
+
|[Robbyant/lingbot-video-dense-1.3b: TI2V](https://modelscope.cn/models/Robbyant/lingbot-video-dense-1.3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-dense-1.3b_ti2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-dense-1.3b_ti2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-dense-1.3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-dense-1.3b_ti2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-dense-1.3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-dense-1.3b_ti2v.py)|
|
|
1522
|
+
|[Robbyant/lingbot-video-dense-1.3b: T2I](https://modelscope.cn/models/Robbyant/lingbot-video-dense-1.3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-dense-1.3b_t2i.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-dense-1.3b_t2i.py)|-|-|-|-|
|
|
1523
|
+
|[Robbyant/lingbot-video-moe-30b-a3b: T2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_t2v.py)|
|
|
1524
|
+
|[Robbyant/lingbot-video-moe-30b-a3b: TI2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_ti2v.py)|
|
|
1525
|
+
|[Robbyant/lingbot-video-moe-30b-a3b: T2I](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2i.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2i.py)|-|-|-|-|
|
|
1526
|
+
|
|
1527
|
+
</details>
|
|
1528
|
+
|
|
1529
|
+
#### MiniMax-H3: [/docs/en/Model_Details/MiniMax-H3.md](/docs/en/Model_Details/MiniMax-H3.md)
|
|
1530
|
+
|
|
1531
|
+
<details>
|
|
1532
|
+
|
|
1533
|
+
<summary>Quick Start</summary>
|
|
1534
|
+
|
|
1535
|
+
Running the following code will quickly load the [DiffSynth-Studio/MiniMax-H3-NF4](https://www.modelscope.cn/models/DiffSynth-Studio/MiniMax-H3-NF4) NF4-quantized model and perform text-to-video-audio inference. VRAM management is enabled, and the framework automatically controls the loading of model parameters based on available VRAM, requiring a minimum of 7GB VRAM.
|
|
1536
|
+
|
|
1537
|
+
```python
|
|
1538
|
+
import torch
|
|
1539
|
+
from diffsynth.pipelines.minimax_h3_audio_video import MiniMaxH3Pipeline, ModelConfig
|
|
1540
|
+
from diffsynth.utils.data.audio_video import write_video_audio
|
|
1541
|
+
|
|
1542
|
+
vram_config = {
|
|
1543
|
+
"offload_dtype": torch.bfloat16,
|
|
1544
|
+
"offload_device": "cpu",
|
|
1545
|
+
"onload_dtype": torch.bfloat16,
|
|
1546
|
+
"onload_device": "cpu",
|
|
1547
|
+
"preparing_dtype": torch.bfloat16,
|
|
1548
|
+
"preparing_device": "cuda",
|
|
1549
|
+
"computation_dtype": torch.bfloat16,
|
|
1550
|
+
"computation_device": "cuda",
|
|
1551
|
+
}
|
|
1552
|
+
pipe = MiniMaxH3Pipeline.from_pretrained(
|
|
1553
|
+
torch_dtype=torch.bfloat16,
|
|
1554
|
+
device="cuda",
|
|
1555
|
+
model_configs=[
|
|
1556
|
+
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-fl2va-nf4.safetensors", **vram_config),
|
|
1557
|
+
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors", **vram_config),
|
|
1558
|
+
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors", **vram_config),
|
|
1559
|
+
ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors", **vram_config),
|
|
1560
|
+
],
|
|
1561
|
+
processor_config=ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="FL2VA/processor/"),
|
|
1562
|
+
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 2,
|
|
1563
|
+
)
|
|
1564
|
+
|
|
1565
|
+
# Text -> Video + Audio
|
|
1566
|
+
prompt = "A girl is very happy, she is speaking in english: “I enjoy working with Diffsynth-Studio, it's a perfect framework.”"
|
|
1567
|
+
video, audio = pipe(
|
|
1568
|
+
prompt=prompt,
|
|
1569
|
+
height=480, width=832, num_frames=124, num_inference_steps=50, seed=0,
|
|
1570
|
+
)
|
|
1571
|
+
write_video_audio(
|
|
1572
|
+
video=video, audio=audio,
|
|
1573
|
+
output_path="t2va.mp4", fps=24, audio_sample_rate=32000,
|
|
1574
|
+
)
|
|
1575
|
+
```
|
|
1576
|
+
|
|
1577
|
+
</details>
|
|
1578
|
+
|
|
1579
|
+
<details>
|
|
1580
|
+
|
|
1581
|
+
<summary>Examples</summary>
|
|
1582
|
+
|
|
1583
|
+
Example code for MiniMax-H3 is available at: [/examples/minimax_h3/](/examples/minimax_h3/)
|
|
1584
|
+
|
|
1585
|
+
| Model ID | Inference | Low VRAM Inference | Full Training | Full Training Validation | LoRA Training | LoRA Training Validation |
|
|
1586
|
+
|-|-|-|-|-|-|-|
|
|
1587
|
+
|[MiniMax/MiniMax-H3: FL2VA](https://www.modelscope.cn/models/MiniMax/MiniMax-H3)|[code](/examples/minimax_h3/model_inference/MiniMax-H3-FL2VA.py)|[code](/examples/minimax_h3/model_inference_low_vram/MiniMax-H3-FL2VA.py)|[code](/examples/minimax_h3/model_training/full/MiniMax-H3-FL2VA.sh)|[code](/examples/minimax_h3/model_training/validate_full/MiniMax-H3-FL2VA.py)|[code](/examples/minimax_h3/model_training/lora/MiniMax-H3-FL2VA.sh)|[code](/examples/minimax_h3/model_training/validate_lora/MiniMax-H3-FL2VA.py)|
|
|
1588
|
+
|[MiniMax/MiniMax-H3: Ref2VA](https://www.modelscope.cn/models/MiniMax/MiniMax-H3)|[code](/examples/minimax_h3/model_inference/MiniMax-H3-Ref2VA.py)|[code](/examples/minimax_h3/model_inference_low_vram/MiniMax-H3-Ref2VA.py)|[code](/examples/minimax_h3/model_training/full/MiniMax-H3-Ref2VA.sh)|[code](/examples/minimax_h3/model_training/validate_full/MiniMax-H3-Ref2VA.py)|[code](/examples/minimax_h3/model_training/lora/MiniMax-H3-Ref2VA.sh)|[code](/examples/minimax_h3/model_training/validate_lora/MiniMax-H3-Ref2VA.py)|
|
|
1589
|
+
|[DiffSynth-Studio/MiniMax-H3-NF4: FL2VA](https://www.modelscope.cn/models/DiffSynth-Studio/MiniMax-H3-NF4)|[code](/examples/minimax_h3/model_inference/MiniMax-H3-NF4-FL2VA.py)|[code](/examples/minimax_h3/model_inference_low_vram/MiniMax-H3-NF4-FL2VA.py)|-|-|[code](/examples/minimax_h3/model_training/lora/MiniMax-H3-NF4-FL2VA.sh)|[code](/examples/minimax_h3/model_training/validate_lora/MiniMax-H3-NF4-FL2VA.py)|
|
|
1590
|
+
|[DiffSynth-Studio/MiniMax-H3-NF4: Ref2VA](https://www.modelscope.cn/models/DiffSynth-Studio/MiniMax-H3-NF4)|[code](/examples/minimax_h3/model_inference/MiniMax-H3-NF4-Ref2VA.py)|[code](/examples/minimax_h3/model_inference_low_vram/MiniMax-H3-NF4-Ref2VA.py)|-|-|[code](/examples/minimax_h3/model_training/lora/MiniMax-H3-NF4-Ref2VA.sh)|[code](/examples/minimax_h3/model_training/validate_lora/MiniMax-H3-NF4-Ref2VA.py)|
|
|
1591
|
+
|
|
1592
|
+
</details>
|
|
1593
|
+
|
|
1450
1594
|
### Audio Synthesis
|
|
1451
1595
|
|
|
1452
1596
|
#### ACE-Step: [/docs/en/Model_Details/ACE-Step.md](/docs/en/Model_Details/ACE-Step.md)
|
|
@@ -1268,14 +1268,37 @@ ideogram4_series = [
|
|
|
1268
1268
|
"model_hash": "6f56a1d28667f2ff98e1c79af88a7516",
|
|
1269
1269
|
"model_name": "ideogram4_dit",
|
|
1270
1270
|
"model_class": "diffsynth.models.ideogram4_dit.Ideogram4DiT",
|
|
1271
|
-
|
|
1271
|
+
# The checkpoint already holds fp8 weights; the backend that reads its layout is
|
|
1272
|
+
# registered next to the model, in `diffsynth.models.ideogram4_dit`.
|
|
1273
|
+
"quant_config": {"method": "ideogram4_fp8", "load_prequantized": True},
|
|
1274
|
+
"extra_kwargs": {"config": {"emb_dim": 4608, "num_layers": 34, "num_heads": 18, "intermediate_size": 12288, "adanln_dim": 512, "in_channels": 128, "llm_features_dim": 53248, "rope_theta": 5000000, "mrope_section": [24, 20, 20], "norm_eps": 1e-05}},
|
|
1272
1275
|
},
|
|
1273
1276
|
{
|
|
1274
1277
|
# Example: ModelConfig(model_id="ideogram-ai/ideogram-4-fp8", origin_file_pattern="text_encoder/model.safetensors")
|
|
1275
1278
|
"model_hash": "6d72a86d1027baff87e2cf8fc523aab1",
|
|
1276
1279
|
"model_name": "ideogram4_text_encoder",
|
|
1277
1280
|
"model_class": "diffsynth.models.ideogram4_text_encoder.Ideogram4TextEncoder",
|
|
1278
|
-
"
|
|
1281
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.ideogram4_text_encoder.Ideogram4TextEncoderStateDictConverter",
|
|
1282
|
+
# Quantized just like the DiT of the same release, and read by the same backend.
|
|
1283
|
+
"quant_config": {"method": "ideogram4_fp8", "load_prequantized": True},
|
|
1284
|
+
},
|
|
1285
|
+
{
|
|
1286
|
+
# Example: ModelConfig(model_id="ideogram-ai/ideogram-4-nf4", origin_file_pattern="transformer/diffusion_pytorch_model.safetensors")
|
|
1287
|
+
# The same file also serves as `unconditional_transformer/diffusion_pytorch_model.safetensors`.
|
|
1288
|
+
"model_hash": "61962d4376a8957f96d807047bc6f780",
|
|
1289
|
+
"model_name": "ideogram4_dit",
|
|
1290
|
+
"model_class": "diffsynth.models.ideogram4_dit.Ideogram4DiT",
|
|
1291
|
+
# bnb's own 4bit layout, written without double quantization.
|
|
1292
|
+
"quant_config": {"method": "bitsandbytes_nf4", "load_prequantized": True, "backend_config_kwargs": {"compress_statistics": False}},
|
|
1293
|
+
"extra_kwargs": {"config": {"emb_dim": 4608, "num_layers": 34, "num_heads": 18, "intermediate_size": 12288, "adanln_dim": 512, "in_channels": 128, "llm_features_dim": 53248, "rope_theta": 5000000, "mrope_section": [24, 20, 20], "norm_eps": 1e-05}},
|
|
1294
|
+
},
|
|
1295
|
+
{
|
|
1296
|
+
# Example: ModelConfig(model_id="ideogram-ai/ideogram-4-nf4", origin_file_pattern="text_encoder/model.safetensors")
|
|
1297
|
+
"model_hash": "c14b8f1328a87ffdba145431277d133a",
|
|
1298
|
+
"model_name": "ideogram4_text_encoder",
|
|
1299
|
+
"model_class": "diffsynth.models.ideogram4_text_encoder.Ideogram4TextEncoder",
|
|
1300
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.ideogram4_text_encoder.Ideogram4TextEncoderStateDictConverter",
|
|
1301
|
+
"quant_config": {"method": "bitsandbytes_nf4", "load_prequantized": True, "backend_config_kwargs": {"compress_statistics": False}},
|
|
1279
1302
|
},
|
|
1280
1303
|
{
|
|
1281
1304
|
# Example: ModelConfig(model_id="DiffSynth-Studio/ideogram-4-bf16-repackage", origin_file_pattern="transformer/diffusion_pytorch_model.safetensors")
|
|
@@ -1289,6 +1312,7 @@ ideogram4_series = [
|
|
|
1289
1312
|
"model_hash": "6a269892c0757aacd46bd41b8d5a7aef",
|
|
1290
1313
|
"model_name": "ideogram4_text_encoder",
|
|
1291
1314
|
"model_class": "diffsynth.models.ideogram4_text_encoder.Ideogram4TextEncoder",
|
|
1315
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.ideogram4_text_encoder.Ideogram4TextEncoderStateDictConverter",
|
|
1292
1316
|
},
|
|
1293
1317
|
]
|
|
1294
1318
|
|
|
@@ -1322,8 +1346,94 @@ krea2_series = [
|
|
|
1322
1346
|
},
|
|
1323
1347
|
]
|
|
1324
1348
|
|
|
1349
|
+
lingbot_video_series = [
|
|
1350
|
+
{
|
|
1351
|
+
# Example: ModelConfig(model_id="Robbyant/lingbot-video-dense-1.3b", origin_file_pattern="transformer/diffusion_pytorch_model.safetensors")
|
|
1352
|
+
"model_hash": "2bcf511fe5e0000519394d242b4d8abd",
|
|
1353
|
+
"model_name": "lingbot_video_dit",
|
|
1354
|
+
"model_class": "diffsynth.models.lingbot_video_dit.LingBotVideoDiT",
|
|
1355
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.lingbot_video_dit.LingBotVideoDiTStateDictConverter",
|
|
1356
|
+
},
|
|
1357
|
+
{
|
|
1358
|
+
# Example: ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors")
|
|
1359
|
+
"model_hash": "65b83aa625cd362ff5ff3409fb367a6f",
|
|
1360
|
+
"model_name": "lingbot_video_dit",
|
|
1361
|
+
"model_class": "diffsynth.models.lingbot_video_dit.LingBotVideoDiT",
|
|
1362
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.lingbot_video_dit.LingBotVideoDiTStateDictConverter",
|
|
1363
|
+
"extra_kwargs": {'depth': 48, 'axes_lens': (4096, 512, 512), 'num_experts': 128, 'moe_intermediate_size': 768, 'n_group': 4, 'topk_group': 2, 'n_shared_experts': 1, 'routed_scaling_factor': 2.5},
|
|
1364
|
+
},
|
|
1365
|
+
]
|
|
1366
|
+
|
|
1367
|
+
minimax_h3_series = [
|
|
1368
|
+
{
|
|
1369
|
+
# Example: ModelConfig(model_id="MiniMax/MiniMax-H3-FL2VA", origin_file_pattern="audio_vae/model.safetensors")
|
|
1370
|
+
"model_hash": "db383f1c8960837b94059f7722e6cb11",
|
|
1371
|
+
"model_name": "minimax_h3_audio_vae",
|
|
1372
|
+
"model_class": "diffsynth.models.minimax_h3_audio_vae.MiniMaxH3AudioVAE",
|
|
1373
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.minimax_h3_audio_vae.MiniMaxH3AudioVAEStateDictConverter"
|
|
1374
|
+
},
|
|
1375
|
+
{
|
|
1376
|
+
# Example: ModelConfig(model_id="MiniMax/MiniMax-H3-FL2VA", origin_file_pattern="text_encoder/model*.safetensors")
|
|
1377
|
+
"model_hash": "8398104725e58fa8287c163c5c332686",
|
|
1378
|
+
"model_name": "minimax_h3_text_encoder",
|
|
1379
|
+
"model_class": "diffsynth.models.minimax_h3_text_encoder.MiniMaxH3TextEncoder",
|
|
1380
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.minimax_h3_text_encoder.MiniMaxH3TextEncoderStateDictConverter",
|
|
1381
|
+
},
|
|
1382
|
+
{
|
|
1383
|
+
# Example: ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-text-encoder-nf4.safetensors")
|
|
1384
|
+
"model_hash": "297933c3a2b0fc4d4dfee30e34c566b8",
|
|
1385
|
+
"model_name": "minimax_h3_text_encoder",
|
|
1386
|
+
"model_class": "diffsynth.models.minimax_h3_text_encoder.MiniMaxH3TextEncoder",
|
|
1387
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.minimax_h3_text_encoder.MiniMaxH3TextEncoderStateDictConverter",
|
|
1388
|
+
"quant_config": {"method": "bitsandbytes_nf4", "load_prequantized": True, "exclude_modules": ["qkv", "proj", "linear_fc1", "linear_fc2"]},
|
|
1389
|
+
},
|
|
1390
|
+
{
|
|
1391
|
+
# Example: ModelConfig(model_id="MiniMax/MiniMax-H3", origin_file_pattern="FL2VA/transformer/model*.safetensors")
|
|
1392
|
+
"model_hash": "db0197b6919425a5c7102c54e73affc1",
|
|
1393
|
+
"model_name": "minimax_h3_dit",
|
|
1394
|
+
"model_class": "diffsynth.models.minimax_h3_dit.MiniMaxH3DiT",
|
|
1395
|
+
},
|
|
1396
|
+
{
|
|
1397
|
+
# Example: ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-fl2va-nf4.safetensors")
|
|
1398
|
+
"model_hash": "4b27efacefbc4d8670e0d7b876699648",
|
|
1399
|
+
"model_name": "minimax_h3_dit",
|
|
1400
|
+
"model_class": "diffsynth.models.minimax_h3_dit.MiniMaxH3DiT",
|
|
1401
|
+
"quant_config": {"method": "bitsandbytes_nf4", "load_prequantized": True, "exclude_modules": ["time_embedder.proj_in", "time_embedder.proj_out", "video_patch_proj", "audio_patch_proj", "condition_proj", "final_layer.video_out", "final_layer.audio_out"]},
|
|
1402
|
+
},
|
|
1403
|
+
{
|
|
1404
|
+
# Example: ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="minimax-h3-ref2va-nf4.safetensors")
|
|
1405
|
+
"model_hash": "ae44c4b18f5eea35d533bf9ca7b529ae",
|
|
1406
|
+
"model_name": "minimax_h3_dit",
|
|
1407
|
+
"model_class": "diffsynth.models.minimax_h3_dit.MiniMaxH3DiT",
|
|
1408
|
+
"quant_config": {"method": "bitsandbytes_nf4", "load_prequantized": True, "exclude_modules": ["time_embedder.proj_in", "time_embedder.proj_out", "video_patch_proj", "audio_patch_proj", "condition_proj", "final_layer.video_out", "final_layer.audio_out"]},
|
|
1409
|
+
},
|
|
1410
|
+
{
|
|
1411
|
+
# Example: ModelConfig(model_id="MiniMax/MiniMax-H3-FL2VA", origin_file_pattern="video_vae/source/model.safetensors")
|
|
1412
|
+
"model_hash": "24b80900992e2024fab17c991c57da23",
|
|
1413
|
+
"model_name": "minimax_h3_video_vae",
|
|
1414
|
+
"model_class": "diffsynth.models.minimax_h3_video_vae.MiniMaxH3VideoVAE",
|
|
1415
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.minimax_h3_video_vae.MiniMaxH3VideoVAEStateDictConverter",
|
|
1416
|
+
},
|
|
1417
|
+
{
|
|
1418
|
+
# Example: ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="video_vae_nf4.safetensors")
|
|
1419
|
+
"model_hash": "204819338f4fa12936684eeda37fdf71",
|
|
1420
|
+
"model_name": "minimax_h3_video_vae",
|
|
1421
|
+
"model_class": "diffsynth.models.minimax_h3_video_vae.MiniMaxH3VideoVAE",
|
|
1422
|
+
"quant_config": {"method": "bitsandbytes_nf4", "load_prequantized": True, "exclude_modules": ["decoder.x_embedder"]},
|
|
1423
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.minimax_h3_video_vae.MiniMaxH3VideoVAEStateDictConverter",
|
|
1424
|
+
},
|
|
1425
|
+
{
|
|
1426
|
+
# Example: ModelConfig(model_id="DiffSynth-Studio/MiniMax-H3-NF4", origin_file_pattern="audio_vae_nf4.safetensors")
|
|
1427
|
+
"model_hash": "925ae768f8f4a9daa902b789de2ad2dc",
|
|
1428
|
+
"model_name": "minimax_h3_audio_vae",
|
|
1429
|
+
"model_class": "diffsynth.models.minimax_h3_audio_vae.MiniMaxH3AudioVAE",
|
|
1430
|
+
"quant_config": {"method": "bitsandbytes_nf4", "load_prequantized": True},
|
|
1431
|
+
"state_dict_converter": "diffsynth.utils.state_dict_converters.minimax_h3_audio_vae.MiniMaxH3AudioVAEStateDictConverter"
|
|
1432
|
+
},
|
|
1433
|
+
]
|
|
1434
|
+
|
|
1325
1435
|
MODEL_CONFIGS = (
|
|
1326
1436
|
stable_diffusion_xl_series + stable_diffusion_series + qwen_image_series + wan_series + flux_series + flux2_series + ernie_image_series
|
|
1327
1437
|
+ z_image_series + ltx2_series + anima_series + mova_series + joyai_image_series + boogu_image_series + ace_step_series + hidream_o1_image_series
|
|
1328
|
-
+ image_metrics_series + ideogram4_series + krea2_series
|
|
1438
|
+
+ image_metrics_series + ideogram4_series + krea2_series + lingbot_video_series + minimax_h3_series
|
|
1329
1439
|
)
|
|
@@ -10,6 +10,45 @@ flux_general_vram_config = {
|
|
|
10
10
|
}
|
|
11
11
|
|
|
12
12
|
VRAM_MANAGEMENT_MODULE_MAPS = {
|
|
13
|
+
"diffsynth.models.minimax_h3_dit.MiniMaxH3DiT": {
|
|
14
|
+
"torch.nn.Linear": "diffsynth.core.vram.layers.AutoWrappedLinear",
|
|
15
|
+
"torch.nn.RMSNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
16
|
+
},
|
|
17
|
+
"diffsynth.models.minimax_h3_text_encoder.MiniMaxH3TextEncoder": {
|
|
18
|
+
"torch.nn.Linear": "diffsynth.core.vram.layers.AutoWrappedLinear",
|
|
19
|
+
"torch.nn.Embedding": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
20
|
+
"torch.nn.LayerNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
21
|
+
"torch.nn.Conv2d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
22
|
+
"torch.nn.Conv3d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
23
|
+
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLTextRMSNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
24
|
+
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLTextRotaryEmbedding": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
25
|
+
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLVisionModel": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
26
|
+
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLVisionPatchEmbed": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
27
|
+
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLVisionRotaryEmbedding": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
28
|
+
},
|
|
29
|
+
"diffsynth.models.minimax_h3_video_vae.MiniMaxH3VideoVAE": {
|
|
30
|
+
"diffsynth.models.minimax_h3_video_vae.WarpedTensor": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
31
|
+
"torch.nn.Linear": "diffsynth.core.vram.layers.AutoWrappedLinear",
|
|
32
|
+
"torch.nn.Embedding": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
33
|
+
"torch.nn.LayerNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
34
|
+
"torch.nn.RMSNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
35
|
+
"torch.nn.Conv3d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
36
|
+
"diffsynth.models.minimax_h3_video_vae.BaseConv3d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
37
|
+
"diffsynth.models.minimax_h3_video_vae.TemporalIsolatedSpatialGroupNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
38
|
+
},
|
|
39
|
+
"diffsynth.models.minimax_h3_audio_vae.MiniMaxH3AudioVAE": {
|
|
40
|
+
"diffsynth.models.minimax_h3_audio_vae.WarpedTensor": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
41
|
+
"torch.nn.Linear": "diffsynth.core.vram.layers.AutoWrappedLinear",
|
|
42
|
+
"torch.nn.LayerNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
43
|
+
"torch.nn.RMSNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
44
|
+
"torch.nn.Conv1d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
45
|
+
"torch.nn.Conv3d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
46
|
+
"diffsynth.models.minimax_h3_audio_vae.Snake": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
47
|
+
"diffsynth.models.minimax_h3_audio_vae.WeightNormedConv1d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
48
|
+
"diffsynth.models.minimax_h3_audio_vae.WeightNormedConvTranspose1d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
49
|
+
"diffsynth.models.minimax_h3_audio_vae.UpSample1d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
50
|
+
"diffsynth.models.minimax_h3_audio_vae.LowPassFilter1d": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
51
|
+
},
|
|
13
52
|
"diffsynth.models.qwen_image_dit.QwenImageDiT": {
|
|
14
53
|
"diffsynth.models.qwen_image_dit.RMSNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
15
54
|
"torch.nn.Linear": "diffsynth.core.vram.layers.AutoWrappedLinear",
|
|
@@ -407,6 +446,16 @@ VRAM_MANAGEMENT_MODULE_MAPS = {
|
|
|
407
446
|
"torch.nn.Embedding": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
408
447
|
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLTextRotaryEmbedding": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
409
448
|
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLTextRMSNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
449
|
+
"torch.nn.LayerNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
450
|
+
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLVisionPatchEmbed": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
451
|
+
"transformers.models.qwen3_vl.modeling_qwen3_vl.Qwen3VLVisionRotaryEmbedding": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
452
|
+
},
|
|
453
|
+
"diffsynth.models.lingbot_video_dit.LingBotVideoDiT": {
|
|
454
|
+
"diffsynth.models.lingbot_video_dit.LingBotVideoBlock": "diffsynth.core.vram.layers.AutoWrappedNonRecurseModule",
|
|
455
|
+
"diffsynth.models.lingbot_video_dit.LingBotVideoRouter": "diffsynth.core.vram.layers.AutoWrappedNonRecurseModule",
|
|
456
|
+
"diffsynth.models.lingbot_video_dit.LingBotVideoGroupedExperts": "diffsynth.core.vram.layers.AutoWrappedNonRecurseModule",
|
|
457
|
+
"diffsynth.models.lingbot_video_dit.LingBotVideoRMSNorm": "diffsynth.core.vram.layers.AutoWrappedModule",
|
|
458
|
+
"torch.nn.Linear": "diffsynth.core.vram.layers.AutoWrappedLinear",
|
|
410
459
|
},
|
|
411
460
|
}
|
|
412
461
|
|
|
@@ -1,9 +1,43 @@
|
|
|
1
|
-
import torch, glob, os
|
|
1
|
+
import torch, glob, os, sys
|
|
2
2
|
from typing import Optional, Union, Dict
|
|
3
3
|
from dataclasses import dataclass
|
|
4
4
|
from modelscope import snapshot_download
|
|
5
5
|
from huggingface_hub import snapshot_download as hf_snapshot_download
|
|
6
6
|
from typing import Optional
|
|
7
|
+
from ..quant import QuantizeConfig, MixedQuantizeConfig
|
|
8
|
+
|
|
9
|
+
_download_tips_printed = False
|
|
10
|
+
|
|
11
|
+
if sys.platform.startswith('win'):
|
|
12
|
+
DOWNLOADING_TIPS = """
|
|
13
|
+
┌──────────────────────────────────────────────────────────────────────┐
|
|
14
|
+
│ DiffSynth-Studio Model Downloader Configuration: │
|
|
15
|
+
│ │
|
|
16
|
+
│ [0] Download from https://modelscope.cn/ │
|
|
17
|
+
│ (default behavior) │
|
|
18
|
+
│ [1] Download from https://modelscope.ai/ │
|
|
19
|
+
│ (enabled via `$env:MODELSCOPE_ENDPOINT="https://modelscope.ai"`) │
|
|
20
|
+
│ [2] Download from https://huggingface.co/ │
|
|
21
|
+
│ (enabled via `$env:DIFFSYNTH_DOWNLOAD_SOURCE="HuggingFace"`) │
|
|
22
|
+
│ [3] Skip download and load only pre-downloaded model files │
|
|
23
|
+
│ (enabled via `$env:DIFFSYNTH_SKIP_DOWNLOAD="True"`) │
|
|
24
|
+
└──────────────────────────────────────────────────────────────────────┘
|
|
25
|
+
""".strip()
|
|
26
|
+
else:
|
|
27
|
+
DOWNLOADING_TIPS = """
|
|
28
|
+
┌──────────────────────────────────────────────────────────────────────┐
|
|
29
|
+
│ DiffSynth-Studio Model Downloader Configuration: │
|
|
30
|
+
│ │
|
|
31
|
+
│ [0] Download from https://modelscope.cn/ │
|
|
32
|
+
│ (default behavior) │
|
|
33
|
+
│ [1] Download from https://modelscope.ai/ │
|
|
34
|
+
│ (enabled via `export MODELSCOPE_ENDPOINT=https://modelscope.ai`) │
|
|
35
|
+
│ [2] Download from https://huggingface.co/ │
|
|
36
|
+
│ (enabled via `export DIFFSYNTH_DOWNLOAD_SOURCE=HuggingFace`) │
|
|
37
|
+
│ [3] Skip download and load only pre-downloaded model files │
|
|
38
|
+
│ (enabled via `export DIFFSYNTH_SKIP_DOWNLOAD=True`) │
|
|
39
|
+
└──────────────────────────────────────────────────────────────────────┘
|
|
40
|
+
""".strip()
|
|
7
41
|
|
|
8
42
|
|
|
9
43
|
@dataclass
|
|
@@ -24,6 +58,7 @@ class ModelConfig:
|
|
|
24
58
|
computation_dtype: Optional[torch.dtype] = None
|
|
25
59
|
clear_parameters: bool = False
|
|
26
60
|
state_dict: Dict[str, torch.Tensor] = None
|
|
61
|
+
quantize: Optional[Union[QuantizeConfig, MixedQuantizeConfig]] = None
|
|
27
62
|
|
|
28
63
|
def check_input(self):
|
|
29
64
|
if self.path is None and self.model_id is None:
|
|
@@ -93,7 +128,30 @@ class ModelConfig:
|
|
|
93
128
|
elif self.local_model_path is None:
|
|
94
129
|
self.local_model_path = "./models"
|
|
95
130
|
|
|
131
|
+
def check_download_source(self):
|
|
132
|
+
download_source = self.parse_download_source().lower()
|
|
133
|
+
if os.environ.get('DIFFSYNTH_SKIP_DOWNLOAD', "").lower() == "true":
|
|
134
|
+
behavior = 3
|
|
135
|
+
elif download_source == "modelscope":
|
|
136
|
+
if "modelscope.ai" in os.environ.get("MODELSCOPE_ENDPOINT", "") or "modelscope.ai" in os.environ.get("MODELSCOPE_DOMAIN", ""):
|
|
137
|
+
behavior = 1
|
|
138
|
+
else:
|
|
139
|
+
behavior = 0
|
|
140
|
+
else:
|
|
141
|
+
behavior = 2
|
|
142
|
+
tips = DOWNLOADING_TIPS
|
|
143
|
+
for i in range(4): tips = tips.replace(f"[{i}]", ["[ ]", "[√]"][i==behavior])
|
|
144
|
+
global _download_tips_printed
|
|
145
|
+
if not _download_tips_printed:
|
|
146
|
+
is_main_process = True
|
|
147
|
+
if torch.distributed.is_initialized():
|
|
148
|
+
is_main_process = torch.distributed.get_rank() == 0
|
|
149
|
+
if is_main_process:
|
|
150
|
+
print(tips)
|
|
151
|
+
_download_tips_printed = True
|
|
152
|
+
|
|
96
153
|
def download_if_necessary(self):
|
|
154
|
+
self.check_download_source()
|
|
97
155
|
self.check_input()
|
|
98
156
|
self.reset_local_model_path()
|
|
99
157
|
if self.require_downloading():
|
|
@@ -33,6 +33,11 @@ def load_state_dict_from_safetensors(file_path, torch_dtype=None, device="cpu"):
|
|
|
33
33
|
return state_dict
|
|
34
34
|
|
|
35
35
|
|
|
36
|
+
def load_metadata_from_safetensors(file_path):
|
|
37
|
+
with safe_open(file_path, framework="pt", device="cpu") as f:
|
|
38
|
+
return f.metadata() or {}
|
|
39
|
+
|
|
40
|
+
|
|
36
41
|
def load_state_dict_from_bin(file_path, torch_dtype=None, device="cpu"):
|
|
37
42
|
state_dict = torch.load(file_path, map_location=device, weights_only=True)
|
|
38
43
|
if len(state_dict) == 1:
|
|
@@ -1,20 +1,20 @@
|
|
|
1
1
|
from ..vram.initialization import skip_model_initialization
|
|
2
2
|
from ..vram.disk_map import DiskMap
|
|
3
3
|
from ..vram.layers import enable_vram_management
|
|
4
|
-
from .file import load_state_dict
|
|
4
|
+
from .file import load_state_dict, load_metadata_from_safetensors
|
|
5
5
|
import torch
|
|
6
6
|
from contextlib import contextmanager
|
|
7
7
|
from transformers.integrations import is_deepspeed_zero3_enabled
|
|
8
8
|
from transformers.utils import ContextManagers
|
|
9
9
|
|
|
10
10
|
|
|
11
|
-
def load_model(model_class, path, config=None, torch_dtype=torch.bfloat16, device="cpu", state_dict_converter=None, use_disk_map=False, module_map=None, vram_config=None, vram_limit=None, state_dict=None):
|
|
11
|
+
def load_model(model_class, path, config=None, torch_dtype=torch.bfloat16, device="cpu", state_dict_converter=None, use_disk_map=False, module_map=None, vram_config=None, vram_limit=None, state_dict=None, quantize=None):
|
|
12
12
|
config = {} if config is None else config
|
|
13
13
|
with ContextManagers(get_init_context(torch_dtype=torch_dtype, device=device)):
|
|
14
14
|
model = model_class(**config)
|
|
15
15
|
# What is `module_map`?
|
|
16
16
|
# This is a module mapping table for VRAM management.
|
|
17
|
-
if module_map is not None:
|
|
17
|
+
if module_map is not None and quantize is None:
|
|
18
18
|
devices = [vram_config["offload_device"], vram_config["onload_device"], vram_config["preparing_device"], vram_config["computation_device"]]
|
|
19
19
|
device = [d for d in devices if d != "disk"][0]
|
|
20
20
|
dtypes = [vram_config["offload_dtype"], vram_config["onload_dtype"], vram_config["preparing_dtype"], vram_config["computation_dtype"]]
|
|
@@ -30,6 +30,66 @@ def load_model(model_class, path, config=None, torch_dtype=torch.bfloat16, devic
|
|
|
30
30
|
else:
|
|
31
31
|
disk_map = DiskMap(path, device, state_dict_converter=state_dict_converter)
|
|
32
32
|
model = enable_vram_management(model, module_map, vram_config=vram_config, disk_map=disk_map, vram_limit=vram_limit)
|
|
33
|
+
elif quantize is not None and module_map is not None:
|
|
34
|
+
if "disk" in vram_config.values():
|
|
35
|
+
if not quantize.load_prequantized:
|
|
36
|
+
raise ValueError("Disk offload with quantization is only supported for pre-quantized checkpoints (load_prequantized=True).")
|
|
37
|
+
devices = [vram_config[k] for k in ("offload_device", "onload_device", "preparing_device", "computation_device")]
|
|
38
|
+
load_device = [d for d in devices if d != "disk"][0]
|
|
39
|
+
disk_map = DiskMap(path, load_device, torch_dtype=None, state_dict_converter=state_dict_converter)
|
|
40
|
+
metadata = load_metadata_from_safetensors(path[0] if isinstance(path, list) else path)
|
|
41
|
+
model = quantize.prepare_for_prequantized_load(model, compute_dtype=vram_config["computation_dtype"])
|
|
42
|
+
model = enable_vram_management(model, module_map, vram_config=vram_config, disk_map=disk_map, vram_limit=vram_limit, quantize=quantize, metadata=metadata)
|
|
43
|
+
else:
|
|
44
|
+
offload_device = vram_config["offload_device"]
|
|
45
|
+
computation_device = vram_config["computation_device"]
|
|
46
|
+
computation_dtype = vram_config["computation_dtype"]
|
|
47
|
+
offload_dtype = vram_config["offload_dtype"]
|
|
48
|
+
load_dtype = None if quantize.load_prequantized else computation_dtype
|
|
49
|
+
if state_dict is None: state_dict = DiskMap(path, offload_device, torch_dtype=load_dtype)
|
|
50
|
+
if state_dict_converter is not None:
|
|
51
|
+
state_dict = state_dict_converter(state_dict)
|
|
52
|
+
else:
|
|
53
|
+
state_dict = {i: state_dict[i] for i in state_dict}
|
|
54
|
+
|
|
55
|
+
if quantize.load_prequantized:
|
|
56
|
+
model = quantize.prepare_for_prequantized_load(model, compute_dtype=computation_dtype)
|
|
57
|
+
state_dict = quantize.unflatten_state_dict(state_dict, load_metadata_from_safetensors(path))
|
|
58
|
+
|
|
59
|
+
model.load_state_dict(state_dict, assign=True)
|
|
60
|
+
state_dict = None
|
|
61
|
+
|
|
62
|
+
model = quantize.quantize_model(model, compute_device=computation_device, model_device=offload_device)
|
|
63
|
+
model = quantize.dequantize_model(model, compute_dtype=computation_dtype, compute_device=computation_device, model_device=offload_device)
|
|
64
|
+
model = model.to(dtype=offload_dtype, device=offload_device)
|
|
65
|
+
model = enable_vram_management(model, module_map, vram_config=vram_config, disk_map=None, vram_limit=vram_limit, quantize=quantize)
|
|
66
|
+
elif quantize is not None:
|
|
67
|
+
# Weight-only quantization (see `diffsynth.core.quant`), isolated from the normal path below.
|
|
68
|
+
if quantize.load_prequantized:
|
|
69
|
+
load_device, load_dtype = device, None
|
|
70
|
+
else:
|
|
71
|
+
load_device, load_dtype = "cpu", torch_dtype
|
|
72
|
+
|
|
73
|
+
if state_dict is not None:
|
|
74
|
+
pass
|
|
75
|
+
elif use_disk_map:
|
|
76
|
+
state_dict = DiskMap(path, load_device, torch_dtype=load_dtype)
|
|
77
|
+
else:
|
|
78
|
+
state_dict = load_state_dict(path, load_dtype, load_device)
|
|
79
|
+
|
|
80
|
+
if state_dict_converter is not None:
|
|
81
|
+
state_dict = state_dict_converter(state_dict)
|
|
82
|
+
else:
|
|
83
|
+
state_dict = {i: state_dict[i] for i in state_dict}
|
|
84
|
+
|
|
85
|
+
if quantize.load_prequantized:
|
|
86
|
+
model = quantize.prepare_for_prequantized_load(model, compute_dtype=torch_dtype or torch.bfloat16)
|
|
87
|
+
state_dict = quantize.unflatten_state_dict(state_dict, load_metadata_from_safetensors(path))
|
|
88
|
+
|
|
89
|
+
model.load_state_dict(state_dict, assign=True)
|
|
90
|
+
model = quantize.quantize_model(model, compute_device=device, model_device=device)
|
|
91
|
+
model = quantize.dequantize_model(model, compute_dtype=torch_dtype or torch.bfloat16)
|
|
92
|
+
model = model.to(dtype=torch_dtype, device=device)
|
|
33
93
|
else:
|
|
34
94
|
# Why do we use `DiskMap`?
|
|
35
95
|
# Sometimes a model file contains multiple models,
|
|
@@ -59,11 +119,7 @@ def load_model(model_class, path, config=None, torch_dtype=torch.bfloat16, devic
|
|
|
59
119
|
# Why do we call `to()`?
|
|
60
120
|
# Because some models override the behavior of `to()`,
|
|
61
121
|
# especially those from libraries like Transformers.
|
|
62
|
-
|
|
63
|
-
# Preserve quantized weights
|
|
64
|
-
model = model.to(dtype=torch_dtype, device=device)
|
|
65
|
-
else:
|
|
66
|
-
model = model.to(device=device)
|
|
122
|
+
model = model.to(dtype=torch_dtype, device=device)
|
|
67
123
|
if hasattr(model, "eval"):
|
|
68
124
|
model = model.eval()
|
|
69
125
|
return model
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from . import bitsandbytes, torchao
|