inline-core 1.2.61__tar.gz → 1.2.63__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {inline_core-1.2.61 → inline_core-1.2.63}/PKG-INFO +19 -12
- {inline_core-1.2.61 → inline_core-1.2.63}/README.md +18 -11
- {inline_core-1.2.61 → inline_core-1.2.63}/pyproject.toml +1 -1
- inline_core-1.2.63/scripts/minimax_h3_lora_check.py +145 -0
- inline_core-1.2.63/scripts/minimax_h3_train_matrix.py +206 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/device/memory.py +3 -2
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/adaln.py +11 -3
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/load.py +61 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/pipeline.py +94 -20
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/requirements.py +55 -9
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/runner.py +8 -4
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/offload.py +4 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/arch.py +75 -8
- inline_core-1.2.63/src/inline_core/training/cache.py +46 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/dataset.py +11 -7
- inline_core-1.2.63/src/inline_core/training/h3.py +425 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/models.py +43 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/trainer.py +23 -19
- inline_core-1.2.63/tests/conftest.py +25 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_extension_install.py +13 -2
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_frontend_serving.py +47 -1
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_minimaxh3_adaln.py +12 -5
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_minimaxh3_nodes.py +50 -3
- inline_core-1.2.63/tests/test_minimaxh3_training.py +361 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/uv.lock +29 -1
- {inline_core-1.2.61 → inline_core-1.2.63}/.gitignore +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/.python-version +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/CLAUDE.md +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/main.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/scripts/flux2_train_matrix.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/scripts/reference.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/components/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/components/conditioning.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/components/interfaces.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/config.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/device/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/device/auto.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/device/detect.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/device/policy.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/device/types.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/errors.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/api.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/constraints.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/fetch.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/handlers.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/importer.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/install.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/loader.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/manifest.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/models.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/paths.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/resolve.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/scanner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/state.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/extensions/tools.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/ffmpeg.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/cache.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/descriptor.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/executor.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/loader_runners.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/primitives.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/registry.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/runners.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/schema.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/topo.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/graph/validate.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/media.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/catalog.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/checkpoint.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/controlspace.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/flux2/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/flux2/controlnet.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/flux2/embeds.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/flux2/provider.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/flux2/requirements.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/flux2/runner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/flux2/variants.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/keymap.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/krea2/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/krea2/convert.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/krea2/depth_control.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/krea2/img2img.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/krea2/provider.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/krea2/requirements.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/krea2/runner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/loaders.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/lora.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/keys.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/provider.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vae_keys.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/autoencoder_kl_minimax_h3.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/autoencoder_kl_minimax_h3_audio.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/before_denoise.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/before_encoder.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/decoders.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/denoise.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/encoders.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/modular_blocks_minimax_h3.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/modular_pipeline.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/packing.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/packing_ref2va.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/scheduling_minimax_h3.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/minimaxh3/vendor/transformer_minimax_h3.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/pipeline_runtime.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/prepared.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/preprocess/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/preprocess/requirements.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/preprocess/runner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/references.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/requirements.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/sampling.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/video_params.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/zimage/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/zimage/primitives.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/zimage/provider.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/zimage/requirements.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/models/zimage/runner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/parallel/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/parallel/config.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/parallel/group.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/parallel/launch.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/parallel/protocol.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/parallel/registry.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/parallel/worker.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/runtime/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/runtime/context.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/runtime/file_store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/runtime/progress.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/runtime/run.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/runtime/store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/runtime/video_encode.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/sampling/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/sampling/batch.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/__main__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/app.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/assets.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/bootstrap.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/frontend.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/manager.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/rpc.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/run_store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/server/serialize.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/assets.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/config.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/fal.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/frames.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/generation.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/graph_build.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/handlers.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/image_meta.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/models.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/moodboard.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/peaks.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/recipe.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/schema.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/system_stats.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/timeline/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/timeline/compose.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/timeline/ffmpeg.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/timeline/render.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/timeline/resolve.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/training.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/studio/training_store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/takes.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/__init__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/__main__.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/caption.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/src/inline_core/training/protocol.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/helpers.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_cache.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_catalog.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_checkpoint.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_config.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_device_detect.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_executor.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_extension_api.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_extension_manifest.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_extension_resolve.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_extension_scanner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_extension_spine.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_extension_state.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_file_store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_flux2_controlnet.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_flux2_folder.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_flux2_resolve.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_flux2_runner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_flux2_training.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_flux2_variants.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_hidden_nodes.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_keymap.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_krea2_convert.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_krea2_depth_control.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_krea2_requirements.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_krea2_runner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_loader_runners.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_loaders.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_lora.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_lora_download.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_memory_policy.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_minimaxh3_keys.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_minimaxh3_load.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_model_requirements.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_offload_prepared.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_output_kind_contract.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_parallel_group.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_pipeline_cache.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_primitives.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_recipe.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_references.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_rpc_bridge.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_run_store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_sampling.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_schema.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_server.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_staged_residency.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_assets.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_fal.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_frames.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_generation.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_graph_build.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_models.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_moodboard.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_multi_reference.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_node_size.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_peaks.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_rpc.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_schema.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_store.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_timeline.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_studio_training.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_take_bytes.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_topo.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_training_arch.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_training_dataset.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_training_models.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_training_resolve.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_validate.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_video_encode.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_video_params.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_webui_install.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_xfuser_sampler.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_zimage_primitives.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_zimage_resolve.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/tests/test_zimage_runner.py +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/webui.bat +0 -0
- {inline_core-1.2.61 → inline_core-1.2.63}/webui.sh +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: inline-core
|
|
3
|
-
Version: 1.2.
|
|
3
|
+
Version: 1.2.63
|
|
4
4
|
Summary: The generation engine behind Inline Studio.
|
|
5
5
|
License-Expression: GPL-3.0-or-later
|
|
6
6
|
Requires-Python: >=3.11
|
|
@@ -68,17 +68,24 @@ The generation engine behind Inline. It takes a typed node graph (JSON) and retu
|
|
|
68
68
|
low-VRAM laptops up to multi-GPU machines that split a single image's sampling across GPUs (via
|
|
69
69
|
xDiT). It is Inline Studio's built-in render backend.
|
|
70
70
|
|
|
71
|
-
Models today: **Z-Image** (Alibaba Tongyi), a 6B rectified-flow
|
|
72
|
-
xDiT already supports, so the multi-GPU split works on it from the
|
|
73
|
-
a 12.9B single-stream MMDiT released as an undistilled RAW base plus an
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
71
|
+
Models today, all running locally on your own GPU: **Z-Image** (Alibaba Tongyi), a 6B rectified-flow
|
|
72
|
+
diffusion transformer and the model xDiT already supports, so the multi-GPU split works on it from the
|
|
73
|
+
start; **Krea 2** (Krea AI), a 12.9B single-stream MMDiT released as an undistilled RAW base plus an
|
|
74
|
+
8-step distilled Turbo; **FLUX.2** (Black Forest Labs); and **MiniMax H3**, which generates video and
|
|
75
|
+
its soundtrack in a single pass.
|
|
76
|
+
|
|
77
|
+
Core also runs the **LoRA trainer** behind Inline Studio's Trainer canvas, so the same engine that
|
|
78
|
+
generates can fine-tune Z-Image, Krea 2, FLUX.2 and MiniMax H3 on your own images without a cloud
|
|
79
|
+
step. Training is cheaper than generating: Krea 2 fits a 16GB card at 512px with a 4-bit base. Its
|
|
80
|
+
dependencies sit behind the `training` extra. See the
|
|
81
|
+
[LoRA training guide](https://inlinestudio.art/lora-training) for the measured VRAM table.
|
|
82
|
+
|
|
83
|
+
> Status: the graph engine, the typed `/v1` HTTP + websocket API (durable runs, streamed progress,
|
|
84
|
+
> coalescing), the model-dir scan, the device + memory policy (profiles, dtype, offload, int8), the
|
|
85
|
+
> low-level primitive node vocabulary, the model runners above, and the LoRA trainer all run on real
|
|
86
|
+
> hardware. Cross-request batching, single-image multi-GPU (an xDiT worker group behind the sampler
|
|
87
|
+
> seam, with the policy and IPC round-trip tested), and out-of-process custom nodes are built as
|
|
88
|
+
> seams but are not yet exercised on real hardware.
|
|
82
89
|
|
|
83
90
|
## Engine design
|
|
84
91
|
|
|
@@ -5,17 +5,24 @@ The generation engine behind Inline. It takes a typed node graph (JSON) and retu
|
|
|
5
5
|
low-VRAM laptops up to multi-GPU machines that split a single image's sampling across GPUs (via
|
|
6
6
|
xDiT). It is Inline Studio's built-in render backend.
|
|
7
7
|
|
|
8
|
-
Models today: **Z-Image** (Alibaba Tongyi), a 6B rectified-flow
|
|
9
|
-
xDiT already supports, so the multi-GPU split works on it from the
|
|
10
|
-
a 12.9B single-stream MMDiT released as an undistilled RAW base plus an
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
8
|
+
Models today, all running locally on your own GPU: **Z-Image** (Alibaba Tongyi), a 6B rectified-flow
|
|
9
|
+
diffusion transformer and the model xDiT already supports, so the multi-GPU split works on it from the
|
|
10
|
+
start; **Krea 2** (Krea AI), a 12.9B single-stream MMDiT released as an undistilled RAW base plus an
|
|
11
|
+
8-step distilled Turbo; **FLUX.2** (Black Forest Labs); and **MiniMax H3**, which generates video and
|
|
12
|
+
its soundtrack in a single pass.
|
|
13
|
+
|
|
14
|
+
Core also runs the **LoRA trainer** behind Inline Studio's Trainer canvas, so the same engine that
|
|
15
|
+
generates can fine-tune Z-Image, Krea 2, FLUX.2 and MiniMax H3 on your own images without a cloud
|
|
16
|
+
step. Training is cheaper than generating: Krea 2 fits a 16GB card at 512px with a 4-bit base. Its
|
|
17
|
+
dependencies sit behind the `training` extra. See the
|
|
18
|
+
[LoRA training guide](https://inlinestudio.art/lora-training) for the measured VRAM table.
|
|
19
|
+
|
|
20
|
+
> Status: the graph engine, the typed `/v1` HTTP + websocket API (durable runs, streamed progress,
|
|
21
|
+
> coalescing), the model-dir scan, the device + memory policy (profiles, dtype, offload, int8), the
|
|
22
|
+
> low-level primitive node vocabulary, the model runners above, and the LoRA trainer all run on real
|
|
23
|
+
> hardware. Cross-request batching, single-image multi-GPU (an xDiT worker group behind the sampler
|
|
24
|
+
> seam, with the policy and IPC round-trip tested), and out-of-process custom nodes are built as
|
|
25
|
+
> seams but are not yet exercised on real hardware.
|
|
19
26
|
|
|
20
27
|
## Engine design
|
|
21
28
|
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Render the same seed with and without a LoRA, and measure what changed.
|
|
2
|
+
|
|
3
|
+
The question a unit test cannot answer: does an adapter trained by the Trainer actually reach the
|
|
4
|
+
denoiser at generation time? A LoRA that changes nothing means the fuse silently missed its
|
|
5
|
+
targets; one that produces noise means the training convention is wrong. Both pass a test suite.
|
|
6
|
+
|
|
7
|
+
Two loads in one process, because the pipeline cache keys on the LoRA stack and evicting between
|
|
8
|
+
them is the same path a user takes when they wire an adapter in.
|
|
9
|
+
|
|
10
|
+
cd core && PYTHONPATH=src .venv/bin/python scripts/minimax_h3_lora_check.py \
|
|
11
|
+
--lora ../outputs/.../skin-h3.safetensors "a close-up portrait ..."
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import json
|
|
18
|
+
import logging
|
|
19
|
+
import sys
|
|
20
|
+
import time
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any
|
|
23
|
+
|
|
24
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger("h3lora")
|
|
27
|
+
OUT = Path(__file__).resolve().parents[2] / "outputs" / "minimax-h3-bench" / "lora-check"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _render(policy: Any, prompt: str, loras: tuple[Any, ...], args: Any) -> dict[str, Any]:
|
|
31
|
+
import torch
|
|
32
|
+
|
|
33
|
+
from inline_core.models import pipeline_runtime as rt
|
|
34
|
+
from inline_core.models.minimaxh3.pipeline import load_pipeline, render_staged
|
|
35
|
+
from inline_core.models.minimaxh3.runner import GRID
|
|
36
|
+
from inline_core.models.video_params import snap_canvas, snap_frames
|
|
37
|
+
|
|
38
|
+
width, height = snap_canvas(args.width, args.height, multiple=32, minimum=32)
|
|
39
|
+
frames = snap_frames(args.seconds, GRID)
|
|
40
|
+
|
|
41
|
+
started = time.perf_counter()
|
|
42
|
+
pipe = load_pipeline(policy, params={}, partition="fl2va", loras=loras)
|
|
43
|
+
load_s = round(time.perf_counter() - started, 1)
|
|
44
|
+
|
|
45
|
+
rt.reset_peak_vram()
|
|
46
|
+
started = time.perf_counter()
|
|
47
|
+
state = render_staged(
|
|
48
|
+
pipe, policy.placement("denoiser").device,
|
|
49
|
+
prompt=prompt, num_frames=frames, height=height, width=width,
|
|
50
|
+
num_inference_steps=args.steps, output_type="pil",
|
|
51
|
+
generator=torch.Generator(device="cpu").manual_seed(args.seed),
|
|
52
|
+
)
|
|
53
|
+
return {
|
|
54
|
+
"load_s": load_s,
|
|
55
|
+
"generate_s": round(time.perf_counter() - started, 1),
|
|
56
|
+
"videos": state.get("videos"),
|
|
57
|
+
"audio": state.get("audio"),
|
|
58
|
+
"sampling_rate": state.get("sampling_rate"),
|
|
59
|
+
"frames": frames,
|
|
60
|
+
"size": (width, height),
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _difference(a: Any, b: Any) -> dict[str, float]:
|
|
65
|
+
"""Mean absolute pixel difference between two clips, in 0-255 units."""
|
|
66
|
+
import numpy as np
|
|
67
|
+
|
|
68
|
+
left = np.stack([np.asarray(f, dtype=np.float32) for f in a])
|
|
69
|
+
right = np.stack([np.asarray(f, dtype=np.float32) for f in b])
|
|
70
|
+
delta = np.abs(left - right)
|
|
71
|
+
return {
|
|
72
|
+
"mean_abs": round(float(delta.mean()), 4),
|
|
73
|
+
"max_abs": round(float(delta.max()), 2),
|
|
74
|
+
"changed_fraction": round(float((delta > 1.0).mean()), 4),
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def main() -> int:
|
|
79
|
+
parser = argparse.ArgumentParser()
|
|
80
|
+
parser.add_argument("prompt")
|
|
81
|
+
parser.add_argument("--lora", required=True)
|
|
82
|
+
parser.add_argument("--strength", type=float, default=1.0)
|
|
83
|
+
parser.add_argument("--label", default="lora-check")
|
|
84
|
+
parser.add_argument("--width", type=int, default=608)
|
|
85
|
+
parser.add_argument("--height", type=int, default=352)
|
|
86
|
+
parser.add_argument("--seconds", type=float, default=5.0)
|
|
87
|
+
parser.add_argument("--steps", type=int, default=20)
|
|
88
|
+
parser.add_argument("--seed", type=int, default=1234)
|
|
89
|
+
args = parser.parse_args()
|
|
90
|
+
|
|
91
|
+
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s", datefmt="%H:%M:%S")
|
|
92
|
+
|
|
93
|
+
from inline_core.device.memory import MemoryPolicy
|
|
94
|
+
from inline_core.graph.loader_runners import LoraRef
|
|
95
|
+
from inline_core.models.minimaxh3.runner import GRID
|
|
96
|
+
from inline_core.runtime.video_encode import encode_video_file
|
|
97
|
+
|
|
98
|
+
out = OUT.parent / args.label
|
|
99
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
100
|
+
policy = MemoryPolicy()
|
|
101
|
+
record: dict[str, Any] = {"prompt": args.prompt, "seed": args.seed, "steps": args.steps}
|
|
102
|
+
|
|
103
|
+
# Base first, so the LoRA'd load is the one that has to evict a live pipeline - which is what a
|
|
104
|
+
# user does when they wire an adapter into a node they have already rendered from.
|
|
105
|
+
logger.info("--- rendering WITHOUT the LoRA ---")
|
|
106
|
+
base = _render(policy, args.prompt, (), args)
|
|
107
|
+
record["without"] = {k: base[k] for k in ("load_s", "generate_s")}
|
|
108
|
+
encode_video_file(
|
|
109
|
+
out / "without-lora.mp4", base["videos"][0], fps=GRID.fps,
|
|
110
|
+
audio=base["audio"][0] if base["audio"] is not None and len(base["audio"]) else None,
|
|
111
|
+
sample_rate=base["sampling_rate"],
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
logger.info("--- rendering WITH the LoRA ---")
|
|
115
|
+
tuned = _render(policy, args.prompt, (LoraRef(file=args.lora, strength=args.strength),), args)
|
|
116
|
+
record["with"] = {k: tuned[k] for k in ("load_s", "generate_s")}
|
|
117
|
+
encode_video_file(
|
|
118
|
+
out / "with-lora.mp4", tuned["videos"][0], fps=GRID.fps,
|
|
119
|
+
audio=tuned["audio"][0] if tuned["audio"] is not None and len(tuned["audio"]) else None,
|
|
120
|
+
sample_rate=tuned["sampling_rate"],
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
record["difference"] = _difference(base["videos"][0], tuned["videos"][0])
|
|
124
|
+
record["lora"] = args.lora
|
|
125
|
+
record["strength"] = args.strength
|
|
126
|
+
|
|
127
|
+
# A few frames side by side, so the change can be looked at rather than only measured.
|
|
128
|
+
for index in (0, base["frames"] // 2, base["frames"] - 1):
|
|
129
|
+
base["videos"][0][index].save(out / f"frame{index:03d}-without.png")
|
|
130
|
+
tuned["videos"][0][index].save(out / f"frame{index:03d}-with.png")
|
|
131
|
+
|
|
132
|
+
(out / "result.json").write_text(json.dumps(record, indent=2))
|
|
133
|
+
print(json.dumps(record, indent=2))
|
|
134
|
+
|
|
135
|
+
delta = record["difference"]["mean_abs"]
|
|
136
|
+
if delta == 0.0:
|
|
137
|
+
print("\nFAIL: identical output - the LoRA reached nothing")
|
|
138
|
+
return 1
|
|
139
|
+
print(f"\nclips differ by {delta} mean absolute (0-255).")
|
|
140
|
+
print("Look at the frames before believing it.")
|
|
141
|
+
return 0
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
if __name__ == "__main__":
|
|
145
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""VRAM + step-time sweep for MiniMax H3 LoRA training: the cells behind the README benchmark table.
|
|
2
|
+
|
|
3
|
+
cd core && PYTHONPATH=src .venv/bin/python scripts/minimax_h3_train_matrix.py --dataset <dir>
|
|
4
|
+
|
|
5
|
+
One cell = one real run of `python -m inline_core.training`, the same entry point the Trainer tab
|
|
6
|
+
spawns, so a number here is a number a user would see.
|
|
7
|
+
|
|
8
|
+
Held fixed at the settings the other architectures' rows used: 12 steps, rank 16, batch 1, gradient
|
|
9
|
+
checkpointing on. Only resolution varies, because H3 has a single base mode and is 4-bit only: its
|
|
10
|
+
base is 40 GB after the AdaLN factorisation, so a bf16 cell would be measuring an OOM.
|
|
11
|
+
|
|
12
|
+
The peak here is not the peak a user waits on. H3 encodes latents and captions in two passes that
|
|
13
|
+
never overlap the base, and the conditioner pass is the tallest of the three, so the run's
|
|
14
|
+
high-water mark is set before training starts. Both are reported.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import argparse
|
|
20
|
+
import json
|
|
21
|
+
import os
|
|
22
|
+
import subprocess
|
|
23
|
+
import time
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
_REPO = Path(__file__).resolve().parent.parent.parent
|
|
27
|
+
_CORE = _REPO / "core"
|
|
28
|
+
_DEFAULT_OUT = _REPO / "outputs" / "minimax-h3-bench" / "train"
|
|
29
|
+
|
|
30
|
+
#: Resolutions to sweep. 512 is the practical setting; 768 matches H3's own short edge at inference.
|
|
31
|
+
CELLS: tuple[int, ...] = (512, 768)
|
|
32
|
+
|
|
33
|
+
_STEPS = 12
|
|
34
|
+
_RANK = 16
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _manifest(
|
|
38
|
+
work: Path, dataset: Path, models: Path, resolution: int, steps: int = _STEPS
|
|
39
|
+
) -> Path:
|
|
40
|
+
"""The same manifest shape `studio/training.py::_prepare` writes."""
|
|
41
|
+
checkpoints = work / "checkpoints"
|
|
42
|
+
checkpoints.mkdir(parents=True, exist_ok=True)
|
|
43
|
+
manifest = {
|
|
44
|
+
"runId": work.name,
|
|
45
|
+
"workingDir": str(work),
|
|
46
|
+
"datasetDir": str(dataset),
|
|
47
|
+
"checkpointDir": str(checkpoints),
|
|
48
|
+
"outputPath": str(work / "lora.safetensors"),
|
|
49
|
+
"resumeFrom": None,
|
|
50
|
+
"modelsDir": str(models),
|
|
51
|
+
"arch": "minimax-h3",
|
|
52
|
+
# H3 ships one undistilled build per partition; `raw` is that mode's key.
|
|
53
|
+
"baseMode": "raw",
|
|
54
|
+
"triggerWord": "",
|
|
55
|
+
"hyperparams": {
|
|
56
|
+
"arch": "minimax-h3",
|
|
57
|
+
"baseMode": "raw",
|
|
58
|
+
"baseQuant": "auto",
|
|
59
|
+
"offload": "off",
|
|
60
|
+
"loraScope": "full",
|
|
61
|
+
"captionDropout": 0.0,
|
|
62
|
+
"flipAugment": False,
|
|
63
|
+
"rank": _RANK,
|
|
64
|
+
"alpha": _RANK,
|
|
65
|
+
"learningRate": 1e-4,
|
|
66
|
+
"batchSize": 1,
|
|
67
|
+
"steps": steps,
|
|
68
|
+
"saveEvery": steps,
|
|
69
|
+
"resolution": resolution,
|
|
70
|
+
},
|
|
71
|
+
"gpuIds": [],
|
|
72
|
+
}
|
|
73
|
+
path = work / "manifest.json"
|
|
74
|
+
path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
|
|
75
|
+
return path
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _run_cell(python: str, manifest: Path, log: Path) -> dict[str, object]:
|
|
79
|
+
"""Drain the JSON-line protocol, keeping the VRAM readings and the step timing.
|
|
80
|
+
|
|
81
|
+
The precache peak is read from the progress lines before the first training step; the training
|
|
82
|
+
peak is the last reading. They are different numbers on H3 and conflating them would overstate
|
|
83
|
+
what training itself costs.
|
|
84
|
+
"""
|
|
85
|
+
env = {**os.environ, "PYTHONPATH": str(_CORE / "src")}
|
|
86
|
+
proc = subprocess.Popen(
|
|
87
|
+
[python, "-m", "inline_core.training", str(manifest)],
|
|
88
|
+
cwd=str(_CORE),
|
|
89
|
+
env=env,
|
|
90
|
+
stdout=subprocess.PIPE,
|
|
91
|
+
stderr=subprocess.STDOUT,
|
|
92
|
+
text=True,
|
|
93
|
+
bufsize=1,
|
|
94
|
+
)
|
|
95
|
+
vram: float | None = None
|
|
96
|
+
error: str | None = None
|
|
97
|
+
first_step_at: float | None = None
|
|
98
|
+
last_step_at: float | None = None
|
|
99
|
+
steps_seen = 0
|
|
100
|
+
losses: list[float] = []
|
|
101
|
+
started = time.perf_counter()
|
|
102
|
+
lines: list[str] = []
|
|
103
|
+
assert proc.stdout is not None
|
|
104
|
+
for line in proc.stdout:
|
|
105
|
+
lines.append(line)
|
|
106
|
+
line = line.strip()
|
|
107
|
+
if not line.startswith("{"):
|
|
108
|
+
continue
|
|
109
|
+
try:
|
|
110
|
+
message = json.loads(line)
|
|
111
|
+
except json.JSONDecodeError:
|
|
112
|
+
continue
|
|
113
|
+
kind = message.get("type")
|
|
114
|
+
if kind == "progress":
|
|
115
|
+
if message.get("vram") is not None:
|
|
116
|
+
vram = float(message["vram"])
|
|
117
|
+
if message.get("loss") is not None:
|
|
118
|
+
losses.append(float(message["loss"]))
|
|
119
|
+
if message.get("step"):
|
|
120
|
+
steps_seen = int(message["step"])
|
|
121
|
+
now = time.perf_counter()
|
|
122
|
+
if first_step_at is None:
|
|
123
|
+
first_step_at = now
|
|
124
|
+
last_step_at = now
|
|
125
|
+
elif kind == "error":
|
|
126
|
+
error = str(message.get("message") or "")
|
|
127
|
+
proc.wait()
|
|
128
|
+
log.write_text("".join(lines), encoding="utf-8")
|
|
129
|
+
|
|
130
|
+
oom = bool(error) and ("out of gpu memory" in error.lower() or "out of memory" in error.lower())
|
|
131
|
+
# Step 1 pays for the first graph build, so time the interval after it and scale by the gap.
|
|
132
|
+
per_step: float | None = None
|
|
133
|
+
if first_step_at is not None and last_step_at is not None and steps_seen > 1:
|
|
134
|
+
per_step = (last_step_at - first_step_at) / (steps_seen - 1)
|
|
135
|
+
return {
|
|
136
|
+
"peak_vram_gb": vram,
|
|
137
|
+
"seconds_per_step": round(per_step, 2) if per_step else None,
|
|
138
|
+
"seconds_12_steps": round(per_step * _STEPS, 1) if per_step else None,
|
|
139
|
+
"total_seconds": round(time.perf_counter() - started, 1),
|
|
140
|
+
"steps_completed": steps_seen,
|
|
141
|
+
"first_loss": round(losses[0], 4) if losses else None,
|
|
142
|
+
"last_loss": round(losses[-1], 4) if losses else None,
|
|
143
|
+
"status": "oom" if oom else ("ok" if proc.returncode == 0 else "failed"),
|
|
144
|
+
"error": error,
|
|
145
|
+
"log": log.name,
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def main() -> int:
|
|
150
|
+
parser = argparse.ArgumentParser()
|
|
151
|
+
parser.add_argument("--dataset", type=Path, required=True, help="dir of NNNN.jpg + NNNN.txt")
|
|
152
|
+
parser.add_argument("--out", type=Path, default=_DEFAULT_OUT)
|
|
153
|
+
parser.add_argument("--models", type=Path, default=_CORE / "models")
|
|
154
|
+
parser.add_argument("--python", default=str(_CORE / ".venv" / "bin" / "python"))
|
|
155
|
+
parser.add_argument("--gpu", default="", help="label for the results file, e.g. 'L40S (46GB)'")
|
|
156
|
+
parser.add_argument("--only", default="", help="one resolution, to redo a single row")
|
|
157
|
+
parser.add_argument("--steps", type=int, default=_STEPS, help="override for a smoke run")
|
|
158
|
+
args = parser.parse_args()
|
|
159
|
+
|
|
160
|
+
if not args.dataset.is_dir():
|
|
161
|
+
raise SystemExit(f"dataset not found: {args.dataset}")
|
|
162
|
+
|
|
163
|
+
args.out.mkdir(parents=True, exist_ok=True)
|
|
164
|
+
results_path = args.out / "results.json"
|
|
165
|
+
results: dict[str, dict[str, object]] = {}
|
|
166
|
+
if results_path.exists():
|
|
167
|
+
results = json.loads(results_path.read_text()).get("cells", {})
|
|
168
|
+
|
|
169
|
+
label = args.gpu
|
|
170
|
+
if not label:
|
|
171
|
+
try:
|
|
172
|
+
label = subprocess.run(
|
|
173
|
+
["nvidia-smi", "--query-gpu=name,memory.total", "--format=csv,noheader"],
|
|
174
|
+
capture_output=True, text=True, check=True,
|
|
175
|
+
).stdout.strip().splitlines()[0]
|
|
176
|
+
except Exception: # noqa: BLE001 - the label is cosmetic
|
|
177
|
+
label = "unknown GPU"
|
|
178
|
+
|
|
179
|
+
def write() -> None:
|
|
180
|
+
results_path.write_text(
|
|
181
|
+
json.dumps(
|
|
182
|
+
{"gpu": label, "steps": _STEPS, "rank": _RANK, "cells": results}, indent=2
|
|
183
|
+
)
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
for resolution in CELLS:
|
|
187
|
+
cell_id = str(resolution)
|
|
188
|
+
if args.only and args.only != cell_id:
|
|
189
|
+
continue
|
|
190
|
+
work = args.out / cell_id
|
|
191
|
+
work.mkdir(parents=True, exist_ok=True)
|
|
192
|
+
manifest = _manifest(work, args.dataset, args.models, resolution, args.steps)
|
|
193
|
+
print(f"--- {cell_id}px, 4-bit base ---", flush=True)
|
|
194
|
+
result = _run_cell(args.python, manifest, args.out / f"{cell_id}.log")
|
|
195
|
+
result.update({"resolution": resolution, "base_quant": "nf4"})
|
|
196
|
+
results[cell_id] = result
|
|
197
|
+
print(json.dumps(result, indent=2), flush=True)
|
|
198
|
+
write()
|
|
199
|
+
|
|
200
|
+
write()
|
|
201
|
+
print(f"\n{results_path}")
|
|
202
|
+
return 0
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
if __name__ == "__main__":
|
|
206
|
+
raise SystemExit(main())
|
|
@@ -302,8 +302,9 @@ class MemoryPolicy(DevicePolicy):
|
|
|
302
302
|
|
|
303
303
|
**int8 overrides the fp16 preference to bf16.** torchao weight-only int8 only supports a
|
|
304
304
|
bf16 compute dtype - with fp16 the quantization silently no-ops, so the "int8" weights load
|
|
305
|
-
at *full* fp16 size and blow the VRAM budget (a T4 then OOMs mid-load). The int8 matmul
|
|
306
|
-
runs on the card's int8 tensor cores; only the residual bf16 activations pay the
|
|
305
|
+
at *full* fp16 size and blow the VRAM budget (a T4 then OOMs mid-load). The int8 matmul
|
|
306
|
+
still runs on the card's int8 tensor cores; only the residual bf16 activations pay the
|
|
307
|
+
slow path.
|
|
307
308
|
bf16 also has fp32's exponent range, so the VAE no longer needs the fp32 anti-overflow
|
|
308
309
|
upcast - it rides along at bf16."""
|
|
309
310
|
if self.quantization() is Quantization.INT8:
|
|
@@ -17,9 +17,17 @@ factorisation from the bf16 weights instead, which means:
|
|
|
17
17
|
needs snapping or interpolation. Projecting a continuous ``t`` through the basis is exact for any
|
|
18
18
|
timestep, so the sampler is unconstrained.
|
|
19
19
|
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
20
|
+
Their tables are **not** compared against; nothing here reads their ``convrot`` build. What is
|
|
21
|
+
measured is this factorisation against the unfactorised weights: it perturbs the modulation by
|
|
22
|
+
1.095e-4 relative, where one bf16 ulp of re-rounding moves it 1.530e-3, so the change sits a factor
|
|
23
|
+
of fourteen below the ambiguity the checkpoint's own storage already carries.
|
|
24
|
+
``scripts/minimax_h3_adaln_gate.py`` computes both and renders the same seed each way; its numbers
|
|
25
|
+
land in ``outputs/minimax-h3-bench/adaln-gate/``.
|
|
26
|
+
|
|
27
|
+
Note the pixel measure disagrees in direction and is not settled: the rendered clips differ by
|
|
28
|
+
0.0385 mean absolute against 0.0258 for that same one-ulp perturbation. Low-rank truncation is
|
|
29
|
+
systematic where rounding is not, so it can compound across the 50 blocks and every step in a way
|
|
30
|
+
the modulation figure does not capture. The gate ran at 8 steps against production's 20.
|
|
23
31
|
|
|
24
32
|
The vendored port is **not** edited for this. The factorised module is swapped in after the model is
|
|
25
33
|
built, so ``vendor/`` stays verbatim.
|
|
@@ -21,6 +21,7 @@ import torch
|
|
|
21
21
|
from safetensors import safe_open
|
|
22
22
|
|
|
23
23
|
from ...errors import ComponentError
|
|
24
|
+
from .. import lora as lora_module
|
|
24
25
|
from ..keymap import (
|
|
25
26
|
AssertEqual,
|
|
26
27
|
Rename,
|
|
@@ -108,12 +109,16 @@ def load_transformer(
|
|
|
108
109
|
device: str = "cpu",
|
|
109
110
|
layout: RowLayout | None = None,
|
|
110
111
|
shrink: Any = None,
|
|
112
|
+
loras: tuple[Any, ...] = (),
|
|
111
113
|
) -> MiniMaxH3Transformer3DModel:
|
|
112
114
|
"""Build the port and stream ``path`` into it through the key plan.
|
|
113
115
|
|
|
114
116
|
``layout`` overrides the measurement, which is only useful in tests; leave it None so the file
|
|
115
117
|
decides. ``shrink(model, prefix)`` is called as each transformer block finishes, which is how a
|
|
116
118
|
64 GB machine loads a model whose bf16 footprint is 66 GB.
|
|
119
|
+
|
|
120
|
+
``loras`` are fused into each block as it lands, **before** ``shrink`` factorises or quantises
|
|
121
|
+
it, because a fuse adds a full-precision delta that quantized weights cannot accept in place.
|
|
117
122
|
"""
|
|
118
123
|
if not path.is_file():
|
|
119
124
|
raise ComponentError(f"MiniMax H3 transformer not found: {path}")
|
|
@@ -121,6 +126,12 @@ def load_transformer(
|
|
|
121
126
|
with torch.device("meta"):
|
|
122
127
|
model = MiniMaxH3Transformer3DModel(**transformer_kwargs(config))
|
|
123
128
|
|
|
129
|
+
# Resolved against module names on the meta model, so a LoRA trained for another architecture
|
|
130
|
+
# is refused before the 62 GB read rather than a block into it.
|
|
131
|
+
lora_plan = lora_module.plan_loras(model, loras) if loras else {}
|
|
132
|
+
fused: set[str] = set()
|
|
133
|
+
shrink = _fusing_shrink(lora_plan, fused, shrink) if lora_plan else shrink
|
|
134
|
+
|
|
124
135
|
with safe_open(str(path), framework="pt") as handle:
|
|
125
136
|
source_keys = list(handle.keys()) # noqa: SIM118 - safe_open has no __contains__
|
|
126
137
|
if _PROBE_KEY not in source_keys:
|
|
@@ -144,11 +155,61 @@ def load_transformer(
|
|
|
144
155
|
model, handle, plan, dtype=dtype, device=device, shrink=shrink
|
|
145
156
|
)
|
|
146
157
|
|
|
158
|
+
if lora_plan:
|
|
159
|
+
_finish_fuse(model, lora_plan, fused)
|
|
147
160
|
_assert_nothing_left_on_meta(model, filled)
|
|
148
161
|
model.eval()
|
|
149
162
|
return model
|
|
150
163
|
|
|
151
164
|
|
|
165
|
+
def _fusing_shrink(plan: Any, fused: set[str], inner: Any) -> Any:
|
|
166
|
+
"""Fuse a block's share of the LoRA stack the moment it lands, then hand off to ``shrink``.
|
|
167
|
+
|
|
168
|
+
The only window that works: after the stream the weights exist, before ``shrink`` they are
|
|
169
|
+
still unquantised, and a full-precision delta cannot be added to a quantised weight.
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
def shrink(model: Any, prefix: str) -> None:
|
|
173
|
+
module = model
|
|
174
|
+
for part in prefix.split("."):
|
|
175
|
+
module = module[int(part)] if part.isdigit() else getattr(module, part)
|
|
176
|
+
fused.update(_fuse_subtree(module, plan, f"{prefix}."))
|
|
177
|
+
if inner is not None:
|
|
178
|
+
inner(model, prefix)
|
|
179
|
+
|
|
180
|
+
return shrink
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _fuse_subtree(module: Any, plan: Any, prefix: str) -> set[str]:
|
|
184
|
+
"""Apply the plan's share for one subtree, reporting which of its targets were covered."""
|
|
185
|
+
hit = {
|
|
186
|
+
path
|
|
187
|
+
for name, _child in module.named_modules()
|
|
188
|
+
if (path := f"{prefix}{name}" if prefix else name) in plan
|
|
189
|
+
}
|
|
190
|
+
lora_module.apply_plan(module, plan, prefix)
|
|
191
|
+
return hit
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _finish_fuse(model: Any, plan: Any, fused: set[str]) -> None:
|
|
195
|
+
"""Fuse what the block callback never saw, then prove nothing was missed.
|
|
196
|
+
|
|
197
|
+
The callback fires only for ``transformer_blocks.N``, so ``context_embedder`` and the token
|
|
198
|
+
refiner would keep their base weights. ``plan_loras`` raises only when a key matches *no*
|
|
199
|
+
module, and these exist, so a partial fuse validates clean and then degrades output silently.
|
|
200
|
+
"""
|
|
201
|
+
residual = {name: deltas for name, deltas in plan.items() if name not in fused}
|
|
202
|
+
if residual:
|
|
203
|
+
fused |= _fuse_subtree(model, residual, "")
|
|
204
|
+
missed = sorted(set(plan) - fused)
|
|
205
|
+
if missed:
|
|
206
|
+
raise ComponentError(
|
|
207
|
+
f"{len(missed)} LoRA layers resolved to modules that were never fused, starting with "
|
|
208
|
+
f"{missed[0]}. Applying only part of a LoRA degrades output without erroring, so this "
|
|
209
|
+
"is refused instead."
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
|
|
152
213
|
def _source_for(layout: RowLayout) -> str:
|
|
153
214
|
for name, known in h3keys.SOURCE_LAYOUTS.items():
|
|
154
215
|
if known is layout:
|