diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/docs/QUANTIZATION.md
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
# Quantization
|
|
2
|
+
|
|
3
|
+
Quantization reduces model memory usage by storing weights at lower precision. diffusers-workflow ships examples for BitsAndBytes, TorchAO, GGUF, and SDNQ, applied per-component in the pipeline - and any other backend with a config class (optimum-quanto, for example) works through the same dynamic `config_type` import.
|
|
4
|
+
|
|
5
|
+
## Per-Component Quantization
|
|
6
|
+
|
|
7
|
+
Quantize individual components (transformer, text encoder, etc.) independently:
|
|
8
|
+
|
|
9
|
+
```json
|
|
10
|
+
{
|
|
11
|
+
"pipeline": {
|
|
12
|
+
"transformer": {
|
|
13
|
+
"configuration": { "component_type": "FluxTransformer2DModel" },
|
|
14
|
+
"quantization_config": {
|
|
15
|
+
"configuration": { "config_type": "..." },
|
|
16
|
+
"arguments": { ... }
|
|
17
|
+
},
|
|
18
|
+
"from_pretrained_arguments": {
|
|
19
|
+
"model_name": "...",
|
|
20
|
+
"subfolder": "transformer",
|
|
21
|
+
"torch_dtype": "torch.bfloat16"
|
|
22
|
+
}
|
|
23
|
+
},
|
|
24
|
+
"configuration": { "component_type": "FluxPipeline" },
|
|
25
|
+
"from_pretrained_arguments": {
|
|
26
|
+
"model_name": "...",
|
|
27
|
+
"torch_dtype": "torch.bfloat16"
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
The component is loaded separately with quantization, then the rest of the pipeline loads around it.
|
|
34
|
+
|
|
35
|
+
## BitsAndBytes (CUDA only)
|
|
36
|
+
|
|
37
|
+
4-bit and 8-bit quantization via bitsandbytes:
|
|
38
|
+
|
|
39
|
+
```json
|
|
40
|
+
"quantization_config": {
|
|
41
|
+
"configuration": { "config_type": "BitsAndBytesConfig" },
|
|
42
|
+
"arguments": {
|
|
43
|
+
"load_in_4bit": true,
|
|
44
|
+
"bnb_4bit_quant_type": "{nf4}",
|
|
45
|
+
"bnb_4bit_compute_dtype": "torch.bfloat16"
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Note: `"{nf4}"` uses braces to keep the string literal. Without braces, the type system would try to load `nf4` as a Python class.
|
|
51
|
+
|
|
52
|
+
For 8-bit:
|
|
53
|
+
|
|
54
|
+
```json
|
|
55
|
+
"arguments": { "load_in_8bit": true }
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
**Example:** [flux2-dev.json](../workflows/models/flux2-dev.json) (a pre-quantized 4-bit checkpoint)
|
|
59
|
+
|
|
60
|
+
## TorchAO
|
|
61
|
+
|
|
62
|
+
Quantization via TorchAO. `quant_type` must be an `AOBaseConfig` class (diffusers no
|
|
63
|
+
longer accepts string shorthands like `"int4wo"`). Name the class as a dotted
|
|
64
|
+
`quant_type` and it is instantiated automatically with no arguments before being passed
|
|
65
|
+
to `TorchAoConfig`:
|
|
66
|
+
|
|
67
|
+
```json
|
|
68
|
+
"quantization_config": {
|
|
69
|
+
"configuration": { "config_type": "TorchAoConfig" },
|
|
70
|
+
"arguments": {
|
|
71
|
+
"quant_type": "torchao.quantization.Int8WeightOnlyConfig",
|
|
72
|
+
"modules_to_not_convert": ["proj_in", "proj_out"]
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Common choices: `Int8WeightOnlyConfig` (any CUDA card), `Int4WeightOnlyConfig` (smallest),
|
|
78
|
+
`Float8DynamicActivationFloat8WeightConfig` (fastest, requires compute capability 8.9+ -
|
|
79
|
+
RTX 40-series/Ada or newer).
|
|
80
|
+
|
|
81
|
+
**Example:** [flux-torchao.json](../workflows/models/flux-torchao.json)
|
|
82
|
+
|
|
83
|
+
**Pair TorchAO with `torch.compile`.** Int8 weight-only and float8 dynamic-activation
|
|
84
|
+
quant types get their fused-kernel speedups only under compilation - uncompiled they are
|
|
85
|
+
a memory win but often a speed *loss*. Compiled is not a guarantee either: int8 weight-only
|
|
86
|
+
on an RTX 3090 measured over a minute per denoising step for Flux dev, compiled, against
|
|
87
|
+
two seconds for bf16 (see [RECIPES_24GB.md](RECIPES_24GB.md#flux-dev-12b)). Measure on the
|
|
88
|
+
card before recording a TorchAO recipe as fast. Add a `compile` block to the quantized component
|
|
89
|
+
(see [ACCELERATION.md](ACCELERATION.md#torchcompile)):
|
|
90
|
+
|
|
91
|
+
```json
|
|
92
|
+
"configuration": {
|
|
93
|
+
"components": {
|
|
94
|
+
"transformer": {
|
|
95
|
+
"compile": { "repeated_blocks": true, "fullgraph": true }
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## GGUF
|
|
102
|
+
|
|
103
|
+
Load GGUF-format checkpoint files:
|
|
104
|
+
|
|
105
|
+
```json
|
|
106
|
+
"quantization_config": {
|
|
107
|
+
"configuration": { "config_type": "GGUFQuantizationConfig" },
|
|
108
|
+
"arguments": {
|
|
109
|
+
"compute_dtype": "torch.bfloat16"
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
GGUF models load from single files using `from_single_file`:
|
|
115
|
+
|
|
116
|
+
```json
|
|
117
|
+
"from_pretrained_arguments": {
|
|
118
|
+
"from_single_file": "https://huggingface.co/city96/FLUX.1-dev-gguf/blob/main/flux1-dev-Q2_K.gguf",
|
|
119
|
+
"torch_dtype": "torch.bfloat16"
|
|
120
|
+
}
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
**Example:** [flux-gguf.json](../workflows/models/flux-gguf.json)
|
|
124
|
+
|
|
125
|
+
## SDNQ (SD.Next Quantization)
|
|
126
|
+
|
|
127
|
+
SDNQ works two ways: quantize a component on the fly at load time, or load a
|
|
128
|
+
pre-quantized model as a complete pipeline.
|
|
129
|
+
|
|
130
|
+
### On-the-fly (`sdnq.SDNQConfig`)
|
|
131
|
+
|
|
132
|
+
Quantizes the component while it loads - the pattern the LTX-2 and MiniMax H3
|
|
133
|
+
examples use for their large transformers and text encoders:
|
|
134
|
+
|
|
135
|
+
```json
|
|
136
|
+
"quantization_config": {
|
|
137
|
+
"configuration": { "config_type": "sdnq.SDNQConfig" },
|
|
138
|
+
"arguments": {
|
|
139
|
+
"weights_dtype": "{uint4}",
|
|
140
|
+
"quantization_device": "cuda",
|
|
141
|
+
"return_device": "cuda",
|
|
142
|
+
"use_quantized_matmul": true,
|
|
143
|
+
"dequantize_fp32": false
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
- `weights_dtype` — the storage dtype (`uint4`, `int8`, ...; brace-escaped so it stays a string)
|
|
149
|
+
- `quantization_device` / `return_device` — where the quantization pass runs and where the finished component lands; quantizing on `cuda` is much faster than on CPU
|
|
150
|
+
- `use_quantized_matmul` — quantized matmul kernels (CUDA/XPU only)
|
|
151
|
+
|
|
152
|
+
**Examples:** [text-to-video.json](../workflows/templates/ltx2/text-to-video.json), [video-with-audio.json](../workflows/templates/minimax/video-with-audio.json)
|
|
153
|
+
|
|
154
|
+
### Pre-quantized models
|
|
155
|
+
|
|
156
|
+
Pre-quantized SDNQ repos load as complete pipelines. The `sdnq` module must be imported before loading so it can register with diffusers:
|
|
157
|
+
|
|
158
|
+
```json
|
|
159
|
+
{
|
|
160
|
+
"pipeline": {
|
|
161
|
+
"configuration": {
|
|
162
|
+
"component_type": "ZImagePipeline",
|
|
163
|
+
"pre_load_modules": ["sdnq"],
|
|
164
|
+
"sdnq_optimize": ["transformer", "text_encoder"]
|
|
165
|
+
},
|
|
166
|
+
"from_pretrained_arguments": {
|
|
167
|
+
"model_name": "Disty0/Z-Image-Turbo-SDNQ-uint4-svd-r32",
|
|
168
|
+
"torch_dtype": "torch.bfloat16"
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
- `pre_load_modules` — Imports sdnq before pipeline loading (registers quantization method)
|
|
175
|
+
- `sdnq_optimize` — Applies quantized matmul to listed components (CUDA/XPU only, skipped on MPS/CPU)
|
|
176
|
+
|
|
177
|
+
**Example:** [z-image-sdnq.json](../workflows/models/z-image-sdnq.json)
|
|
178
|
+
|
|
179
|
+
## Modular Pipelines
|
|
180
|
+
|
|
181
|
+
A modular pipeline pulls its component weights itself via `load_components()` rather than
|
|
182
|
+
through `from_pretrained_arguments`, so quantization is keyed by component name under
|
|
183
|
+
`load_components.quantization_config` instead of living on a separate component block:
|
|
184
|
+
|
|
185
|
+
```json
|
|
186
|
+
"configuration": {
|
|
187
|
+
"component_type": "MiniMaxMusic3ModularPipeline",
|
|
188
|
+
"load_components": {
|
|
189
|
+
"dtype": "torch.bfloat16",
|
|
190
|
+
"quantization_config": {
|
|
191
|
+
"transformer": {
|
|
192
|
+
"configuration": { "config_type": "TorchAoConfig" },
|
|
193
|
+
"arguments": { "quant_type": "torchao.quantization.Int8WeightOnlyConfig" }
|
|
194
|
+
},
|
|
195
|
+
"language_model": {
|
|
196
|
+
"configuration": { "config_type": "transformers.TorchAoConfig" },
|
|
197
|
+
"arguments": { "quant_type": "torchao.quantization.Int8WeightOnlyConfig" }
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
A component the map does not name loads unquantized. Note that a transformers-based
|
|
205
|
+
component (a text encoder, for example) takes the `transformers.TorchAoConfig` class,
|
|
206
|
+
not the diffusers one - the `config_type` still resolves either via the dynamic import
|
|
207
|
+
described below.
|
|
208
|
+
|
|
209
|
+
## Custom Quantization
|
|
210
|
+
|
|
211
|
+
Any quantization backend that provides a config class works via the `config_type` field with a dotted module path:
|
|
212
|
+
|
|
213
|
+
```json
|
|
214
|
+
"quantization_config": {
|
|
215
|
+
"configuration": { "config_type": "some_package.SomeQuantConfig" },
|
|
216
|
+
"arguments": { ... }
|
|
217
|
+
}
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
The class is loaded dynamically via importlib.
|
|
221
|
+
|
|
222
|
+
## Platform Notes
|
|
223
|
+
|
|
224
|
+
| Framework | CUDA | MPS | CPU |
|
|
225
|
+
| --------- | ---- | --- | --- |
|
|
226
|
+
| BitsAndBytes | Yes | No | No |
|
|
227
|
+
| TorchAO | Yes | Partial | No |
|
|
228
|
+
| GGUF | Yes | Yes | Yes |
|
|
229
|
+
| SDNQ (load) | Yes | Yes | Yes |
|
|
230
|
+
| SDNQ (optimize) | Yes | No | No |
|
dw/docs/RECIPES_24GB.md
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# Fast on 24GB
|
|
2
|
+
|
|
3
|
+
Recommended configurations per model family for a 24GB consumer GPU (RTX 3090/4090 class). Every knob here is documented individually in [ACCELERATION.md](ACCELERATION.md) and [QUANTIZATION.md](QUANTIZATION.md); this page is about the combinations that work.
|
|
4
|
+
|
|
5
|
+
The general recipe, in order of impact:
|
|
6
|
+
|
|
7
|
+
1. **Fit the transformer first.** If it fits in bf16 with room for activations, don't quantize. If it doesn't, prefer float8/int8 quantization (TorchAO, GGUF Q8) over offloading - quantization costs quality once, offloading costs speed every step.
|
|
8
|
+
2. **Compile the transformer** (`"compile": {"repeated_blocks": true}`). 1.3-1.5x, stacks with everything below. The REPL worker keeps compiled pipelines loaded, so the compile cost is paid once per session. Add `fullgraph: true` only when no cache is configured - cache hooks need a graph break.
|
|
9
|
+
3. **Cache** (`"cache": {"type": "first_block"}`). Another 1.5-2x at mild quality cost; raise `threshold` to taste.
|
|
10
|
+
4. **Offload only what doesn't fit.** Text encoders and VAE tolerate `offload: "model"` cheaply - they run once per generation, not once per step. In a modular pipeline the same components take `"residency": "on_demand"`, which frees their VRAM for the denoise loop at the cost of one pair of transfers per call.
|
|
11
|
+
5. **Pin the attention backend** on compiled components (`"attention_backend": "flash_hub"` or `"sage_hub"` - fetched from the Hub, no local build).
|
|
12
|
+
|
|
13
|
+
## Flux dev (12B)
|
|
14
|
+
|
|
15
|
+
The bf16 transformer is ~24GB - it does not fit alongside the T5 encoder, and it does
|
|
16
|
+
not fit alongside the VAE either: a resident bf16 transformer runs the denoise and then
|
|
17
|
+
fails at decode. `offload: "model"` is what makes bf16 work on 24GB, and on an RTX 3090
|
|
18
|
+
it is also the fastest configuration measured (1024x1024, 28 steps, pipeline loaded):
|
|
19
|
+
|
|
20
|
+
| Approach | Config | Measured on RTX 3090 |
|
|
21
|
+
| -------- | ------ | -------------------- |
|
|
22
|
+
| **bf16 + model offload** | [flux-dev.json](../workflows/models/flux-dev.json) | 55s per image (72s cold) |
|
|
23
|
+
| **bf16 + model offload + compile** | [flux-dev-compile.json](../workflows/models/flux-dev-compile.json) - `compile: {repeated_blocks: true}` on the transformer | 52s per image (64s cold) |
|
|
24
|
+
| **int8 TorchAO** | transformer `quant_type: "torchao.quantization.Int8WeightOnlyConfig"` + `compile`, with or without `cache: first_block` | over 60s per *denoising step* - do not use on Ampere |
|
|
25
|
+
| **float8 TorchAO** (RTX 40-series+) | `Float8DynamicActivationFloat8WeightConfig` + `compile` | needs compute capability 8.9+ (Ada); unmeasured |
|
|
26
|
+
| **GGUF Q8** | [flux-gguf.json](../workflows/models/flux-gguf.json) - `from_single_file` Q8_0 transformer + `offload: "model"` | unmeasured; smallest VRAM, best quality retention |
|
|
27
|
+
|
|
28
|
+
Measured 2026-09-07. A number in this table came from a run on the named card; a row
|
|
29
|
+
without one is a configuration that loads, not a speed claim.
|
|
30
|
+
|
|
31
|
+
**Examples:** [flux-dev-compile.json](../workflows/models/flux-dev-compile.json), [flux-gguf.json](../workflows/models/flux-gguf.json), [step-caching.json](../workflows/templates/step-caching.json)
|
|
32
|
+
|
|
33
|
+
## Qwen-Image (20B)
|
|
34
|
+
|
|
35
|
+
Too large for bf16 on 24GB. Quantize the transformer (TorchAO int8/float8 or GGUF Q4/Q5) and model-offload the rest; add `compile` + `first_block` cache as for Flux. The Qwen2.5-VL text encoder is large - quantize it too, or group-offload it (`"text_encoder.model"` with `leaf_level`).
|
|
36
|
+
|
|
37
|
+
The catalog no longer ships a Qwen-Image workflow - author one from [text-to-image.json](../workflows/templates/text-to-image.json) with the configuration above.
|
|
38
|
+
|
|
39
|
+
## Wan 2.2
|
|
40
|
+
|
|
41
|
+
- **TI2V-5B**: fits in bf16 on 24GB. No quantization needed - just `compile` + `cache` + VAE tiling for longer clips.
|
|
42
|
+
- **T2V/I2V-A14B**: two 14B transformers (`transformer` + `transformer_2`). Quantize both (GGUF Q4/Q5 or TorchAO int8) and use `offload: "model"`; enable `vae.enable_tiling`.
|
|
43
|
+
|
|
44
|
+
The catalog no longer ships Wan workflows - author one from [minimax/image-to-video.json](../workflows/templates/minimax/image-to-video.json) with the configuration above.
|
|
45
|
+
|
|
46
|
+
## HunyuanVideo (13B)
|
|
47
|
+
|
|
48
|
+
Same shape as Flux: quantize the transformer (GGUF Q6/Q8, or TorchAO int8/float8 per the Flux table) + `offload: "model"` + `vae.enable_tiling`. The Llama text encoder benefits from group offload. Use `first_block` cache - video steps are expensive, caching pays off more than on images.
|
|
49
|
+
|
|
50
|
+
The catalog no longer ships a HunyuanVideo workflow - author one from [ltx2/text-to-video.json](../workflows/templates/ltx2/text-to-video.json) with the configuration above.
|
|
51
|
+
|
|
52
|
+
## MiniMax-H3
|
|
53
|
+
|
|
54
|
+
A modular pipeline, so everything is per component in `components` rather than a
|
|
55
|
+
pipeline-level `offload`. The working 24GB configuration at 960x544:
|
|
56
|
+
|
|
57
|
+
| Component | Config |
|
|
58
|
+
| --------- | ------ |
|
|
59
|
+
| `transformer` / `transformer_ref` | SDNQ int4 (`quantization_device: "cuda"`, `return_device: "cpu"`), `group_offload` `block_level` with `num_blocks_per_group: 1-2` and `use_stream: true` |
|
|
60
|
+
| `text_encoder` | `remove_modules: ["lm_head"]` - the encoder path never calls it |
|
|
61
|
+
| `text_encoder.model` | SDNQ int4, `truncate_layers: {"language_model.layers": 51}`, `group_offload` `leaf_level` |
|
|
62
|
+
| `vae`, `audio_vae` | SDNQ int8, `device: "cuda"`, `residency: "on_demand"` |
|
|
63
|
+
| pipeline | `cache: first_block` (`threshold: 0.1`) on the 20-step ref2va workflows only - measured on the 9-step turbo schedule it never skips (consecutive distilled steps differ too much for the threshold), so the turbo examples omit it rather than hold cache state for nothing |
|
|
64
|
+
|
|
65
|
+
The VAEs are the piece worth calling out. They hold roughly 3GiB, are used only to
|
|
66
|
+
encode references and decode the result, and group offloading them is worse than useless
|
|
67
|
+
because tiled decode restreams the model once per tile. Every H3 workflow adds
|
|
68
|
+
`"residency": "on_demand"` to both. The reference workflows, which carry the most
|
|
69
|
+
conditioning, go from 23.2GiB peak reserved with 40 allocator retries to 18.9GiB with
|
|
70
|
+
none; the frame-conditioned ones from 22.7GiB with 22 retries to 18.0GiB with none. Both
|
|
71
|
+
for about 1% in wall time.
|
|
72
|
+
Spend the headroom on length: carrying a frame between chained segments adds a reference
|
|
73
|
+
and ~1.9GiB, which is what made the chained variants OOM on their second segment before.
|
|
74
|
+
|
|
75
|
+
The text encoder pruning exists because H3 conditions on `hidden_states[50]` of its
|
|
76
|
+
64-layer Qwen3-VL: layers 51-63 and the LM head run (and stream) for nothing on every
|
|
77
|
+
encode. Keeping 51 layers is bit-identical - index 50 of the tuple is recorded before
|
|
78
|
+
layer 50 runs; keeping only 50 would hand back the final-norm output, a different
|
|
79
|
+
tensor - and it returns a few GiB of system RAM on a host that needs every one of them
|
|
80
|
+
(a full t2va run peaks around 63GiB RSS on a 64GiB box). Note H3's VAE constructs with
|
|
81
|
+
tiling already enabled, so a pipeline-level `vae.enable_tiling` adds nothing here.
|
|
82
|
+
|
|
83
|
+
Two levers measured and *rejected* on a 3090 (A/B, t2va 960x544x124f, 2026-08):
|
|
84
|
+
`low_cpu_mem_usage: false` on the transformer's group offload (pin host copies once
|
|
85
|
+
instead of re-pinning per onload) left step time unchanged at ~15.4s - at int4 the
|
|
86
|
+
step is compute-bound and the transfers already hide under it - while the ~27GiB of
|
|
87
|
+
unswappable pinned memory pushed the host into an OOM kill. `use_stream` on the text
|
|
88
|
+
encoder's leaf offload was backed out with it for the same host-memory reason. On a
|
|
89
|
+
faster GPU (or a host with more RAM) both are worth re-testing; the step budget there
|
|
90
|
+
may actually expose the transfer time.
|
|
91
|
+
|
|
92
|
+
Length costs VRAM but the configuration holds to the model's full range: a single
|
|
93
|
+
345-frame take (14.4s, the `17n+5` maximum) peaks at 23.6GiB reserved at 960x544 -
|
|
94
|
+
inside 24GB with nothing to spare - and denoises in ~13 minutes on a 3090 with the
|
|
95
|
+
9-step turbo schedule (~85-100s a step once warm, against ~15s at 124 frames).
|
|
96
|
+
|
|
97
|
+
Host RAM is the tighter budget than VRAM on a 64GiB box. Loading H3 peaks around
|
|
98
|
+
59GiB RSS and a running ref2va shot sits at 45-53GiB, so a workflow that ran another
|
|
99
|
+
model first (Z-Image drawing a subject, Music3 writing a song) must free it with
|
|
100
|
+
`release_pipeline` before H3 loads - with it, the multi-model digital-short
|
|
101
|
+
workflows below fit; without it, the load is an OOM kill, not a slowdown.
|
|
102
|
+
|
|
103
|
+
**Examples:** [reference-to-video.json](../workflows/templates/minimax/reference-to-video.json), [chain-matched-to-audio.json](../workflows/templates/minimax/chain-matched-to-audio.json), [image-to-video.json](../workflows/templates/minimax/image-to-video.json), [dialogue-short.json](../workflows/templates/minimax/dialogue-short.json) (five ref2va shots + two Z-Image portraits in ~35 minutes end to end)
|
|
104
|
+
|
|
105
|
+
## MiniMax-Music3
|
|
106
|
+
|
|
107
|
+
Music3 runs at about 22GiB in bfloat16 under the templates' `components_manager`
|
|
108
|
+
auto CPU offload, which keeps only the running component resident; no quantization
|
|
109
|
+
is needed on a 24GB card. The language model is the part worth offloading harder:
|
|
110
|
+
a leaf-level `group_offload` of `language_model` brings it to about 8GiB (the model
|
|
111
|
+
card's low-VRAM recipe). Two things the examples carry: `release_pipeline` on the
|
|
112
|
+
music step in any workflow that loads H3 afterwards, since host RAM is the binding
|
|
113
|
+
constraint (see Multi-model workflows below), and the run's time follows the length
|
|
114
|
+
the model actually sings, not `audio_duration`, which is a ceiling of at most 9000
|
|
115
|
+
frames at 25 frames per second (360 seconds). Output is 44.1 kHz stereo.
|
|
116
|
+
|
|
117
|
+
**Examples:** [music.json](../workflows/templates/minimax/music.json), [music-video.json](../workflows/templates/minimax/music-video.json)
|
|
118
|
+
|
|
119
|
+
## LTX-2.5 (22B, video + audio)
|
|
120
|
+
|
|
121
|
+
A standard pipeline, but placed per component rather than with a pipeline-level
|
|
122
|
+
`offload` - the transformer is the only thing that wants to be resident, and the text
|
|
123
|
+
encoder is nearly as large as it is. The working 24GB configuration at 960x544:
|
|
124
|
+
|
|
125
|
+
| Component | Config |
|
|
126
|
+
| --------- | ------ |
|
|
127
|
+
| `transformer` | SDNQ `uint4` (`quantization_device: "cuda"`, `return_device: "cuda"`, `use_quantized_matmul: true`), resident |
|
|
128
|
+
| `text_encoder` (Gemma 4, 23GB) | SDNQ `int8`, `return_device: "cpu"`, `group_offload` `leaf_level` with `use_stream: true` |
|
|
129
|
+
| `connectors` (12GB) | `group_offload` `leaf_level` with `use_stream: true` |
|
|
130
|
+
| `vae`, `audio_vae`, `vocoder`, `duration_head` | `device: "cuda"` - small, and used once per generation |
|
|
131
|
+
| pipeline | `vae.enable_tiling` for anything above the base resolution |
|
|
132
|
+
|
|
133
|
+
Two things about the checkpoint are worth knowing before tuning anything:
|
|
134
|
+
|
|
135
|
+
- **`transformer` is the distilled model.** It runs a fixed 8-step schedule at
|
|
136
|
+
`guidance_scale: 1.0`, with STG and modality guidance off, and the `sigmas` every
|
|
137
|
+
example passes are its trained schedule - not a knob. They are referenced from
|
|
138
|
+
diffusers (`constant:diffusers.pipelines.ltx2.utils.DISTILLED_SIGMA_VALUES`) rather
|
|
139
|
+
than copied, so the schedule stays whatever the library says it is. `num_inference_steps`,
|
|
140
|
+
`guidance_scale`, `stg_scale` and the rest only mean anything against
|
|
141
|
+
`subfolder: "transformer_full"`, the dev model, which is not a 24GB configuration:
|
|
142
|
+
it is the same ~38GB in bf16, and the guidance those knobs turn on costs three
|
|
143
|
+
transformer passes per step against CFG-doubled batches. Nothing here ships it.
|
|
144
|
+
- **The checkpoint ships a diffusion decoder that `LTX2Pipeline` ignores**, and on
|
|
145
|
+
24GB you are not missing much. It is listed in `model_index.json` but is not a
|
|
146
|
+
constructor argument, so diffusers logs "not expected ... will be ignored" and
|
|
147
|
+
decodes with the convolutional VAE. Reaching it means `LTX2VideoDiffusionDecodePipeline`
|
|
148
|
+
on a step run with `output_type: "{latent}"`, and two things get in the way. Its
|
|
149
|
+
neighborhood attention has two processors, and the one you get by default is the
|
|
150
|
+
portable FlexAttention fallback: it densifies a `seq_len x seq_len` block mask and then
|
|
151
|
+
runs uncompiled `flex_attention`, which falls to the eager reference path. Both
|
|
152
|
+
allocations are quadratic in the output grid, and neither is reduced by tiling (stages
|
|
153
|
+
1-3 always run on the full volume) or by shrinking the clip (the stage-4 grid is near
|
|
154
|
+
output resolution either way). Measured on an otherwise empty 3090 (#153): 10.05GiB
|
|
155
|
+
inside stages 1-3 at 960x544x121, 69.77GiB for one stage-4 attention at 512x288x25
|
|
156
|
+
(17.44GiB just to densify that stage's mask, whichever it reaches first), and ~25.5GiB
|
|
157
|
+
at 224x224x25, which is the smallest canvas its 7x7 kernel accepts at all. Nothing
|
|
158
|
+
fits - not base resolution, not the smallest clip the decoder will take. The path that does is NATTEN's `na3d` kernel, named per component as
|
|
159
|
+
`"attn_processor_type": "diffusers.models.autoencoders.ltx2_diffusion_decoder.LTX2VideoVaeNeighborhoodNattenProcessor"`,
|
|
160
|
+
which builds no mask at all - but it is fetched from the Hub by the `kernels` package
|
|
161
|
+
and needs a `shi-labs/natten` build matching the installed torch, which as of
|
|
162
|
+
torch 2.14 does not exist.
|
|
163
|
+
And a step that returns latents returns *audio* latents too, which nothing outside a
|
|
164
|
+
pipeline call can vocode - the two-stage template below feeds them back into one,
|
|
165
|
+
which is the only way they become sound. Nothing here ships the diffusion decoder.
|
|
166
|
+
|
|
167
|
+
Spend headroom on the two-stage flow rather than on base resolution: render at 768x448
|
|
168
|
+
on the eight distilled sigmas, double the video latents with the latent upsampler, then
|
|
169
|
+
renoise them and run the three stage-two sigmas at 1536x896 through the same pipeline.
|
|
170
|
+
That refine pass is what puts the detail back - the upsampler alone gives a soft 2x -
|
|
171
|
+
and it is the flow the model card, Lightricks' pipeline notes and the diffusers docs all
|
|
172
|
+
describe. The base pass keeps its pipeline loaded so the refine pass is served from the
|
|
173
|
+
cache; the refine pass releases it. Since 2026-08 Lightricks route production quality
|
|
174
|
+
through their DFR pipeline instead, which diffusers ships and nothing here uses yet.
|
|
175
|
+
Measured on an RTX 3090 the refined clip is sharper than the 2x upsample alone at the
|
|
176
|
+
same seed: fur, branches and snow texture resolve where the upsample-only frame is a
|
|
177
|
+
soft blur. About eight warm minutes, three and a half of them writing the full-size clip.
|
|
178
|
+
|
|
179
|
+
**Examples:** [text-to-video.json](../workflows/templates/ltx2/text-to-video.json) (t2v),
|
|
180
|
+
[two-stage.json](../workflows/templates/ltx2/two-stage.json) (base -> latent upsample -> refine),
|
|
181
|
+
[keyframes.json](../workflows/templates/ltx2/keyframes.json) (first and
|
|
182
|
+
last frame), [extend-clip.json](../workflows/templates/ltx2/extend-clip.json) (continue a clip),
|
|
183
|
+
[generative-upscale.json](../workflows/templates/ltx2/generative-upscale.json) (generative 2x upscale via IC-LoRA),
|
|
184
|
+
[enhance-prompt.json](../workflows/templates/ltx2/enhance-prompt.json) (native prompt
|
|
185
|
+
enhancer and duration head)
|
|
186
|
+
|
|
187
|
+
## SDXL (2.6B UNet)
|
|
188
|
+
|
|
189
|
+
Fits several times over in 24GB. Skip quantization and offloading entirely; `compile` the UNet if you generate many images per session. Use `num_images_per_prompt` batching with `vae.enable_slicing`.
|
|
190
|
+
|
|
191
|
+
**Example:** [base-and-refiner.json](../workflows/templates/base-and-refiner.json)
|
|
192
|
+
|
|
193
|
+
## Multi-model workflows
|
|
194
|
+
|
|
195
|
+
When a workflow chains two large models (generate → upscale, generate → interpolate), release the first pipeline instead of offloading everything:
|
|
196
|
+
|
|
197
|
+
```json
|
|
198
|
+
{ "name": "generate", "release_pipeline": true, "pipeline": { ... } }
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
See [WORKFLOW_GUIDE.md](WORKFLOW_GUIDE.md#releasing-a-pipeline-mid-workflow).
|
dw/docs/RELEASING.md
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
# Releasing
|
|
2
|
+
|
|
3
|
+
## Unreleased
|
|
4
|
+
|
|
5
|
+
This project has no standing release-notes file - GitHub auto-generates
|
|
6
|
+
notes from commits at tag time (see below). This section is a scratch pad
|
|
7
|
+
for items a branch's author wants the next release note to name; clear it
|
|
8
|
+
when a release ships.
|
|
9
|
+
|
|
10
|
+
### 0.4.0
|
|
11
|
+
|
|
12
|
+
The auto-generated notes for this range are a single merge line, since the
|
|
13
|
+
work landed on `develop` without PRs. Paste this section into the GitHub
|
|
14
|
+
release body once the tag has published (`gh release edit v0.4.0
|
|
15
|
+
--notes-file ...`).
|
|
16
|
+
|
|
17
|
+
**Breaking and behaviour changes**
|
|
18
|
+
|
|
19
|
+
- `download_output` over a `dw.serve --mcp` endpoint refuses a call with no
|
|
20
|
+
`destination`. It used to write into the server's own directory (#353).
|
|
21
|
+
- Untrusted workflows are refused in more cases (#409-#413):
|
|
22
|
+
- a `*_type` that doesn't resolve to a class, or that isn't a kind a
|
|
23
|
+
workflow constructs: a diffusers or transformers model, pipeline,
|
|
24
|
+
scheduler, tokenizer or processor, a quantization config, an auto
|
|
25
|
+
factory, a diffusers reference/condition type or an attention processor.
|
|
26
|
+
A plain `torch` class such as `torch.nn.Linear` is now refused;
|
|
27
|
+
- `constant:` walks through `_` names or out of the allowed packages;
|
|
28
|
+
- URLs with backslashes;
|
|
29
|
+
- `text/html` and `text/xml` result types;
|
|
30
|
+
- media hosts that aren't globally routable, including 100.64/10 (CGNAT,
|
|
31
|
+
and so Tailscale);
|
|
32
|
+
- more than 5 redirects;
|
|
33
|
+
- images over 50M pixels.
|
|
34
|
+
|
|
35
|
+
Listings and export zips drop symlinks that escape their root.
|
|
36
|
+
`--trust-workflows` lifts all of these.
|
|
37
|
+
- `run_workflow` validates the caller's `arguments` when it queues the job
|
|
38
|
+
(#414/#415). `validate_workflow(arguments={})` checks a run with no values
|
|
39
|
+
supplied, not just the document (#364).
|
|
40
|
+
- A fractional value for an int variable is refused (#338), and so is a
|
|
41
|
+
still image passed as a video argument (#347).
|
|
42
|
+
- `templates/minimax/music` normalizes to -3 dBFS instead of -1, so its output
|
|
43
|
+
is quieter (#362).
|
|
44
|
+
- Every response carries `X-Content-Type-Options: nosniff` and
|
|
45
|
+
`X-Frame-Options: DENY`. Active document types under `/outputs` and
|
|
46
|
+
`/inputs` are served with `Content-Security-Policy: sandbox`.
|
|
47
|
+
- A validate-time probe reads only a literal media path that the run itself
|
|
48
|
+
would be allowed to read.
|
|
49
|
+
- A dict or list passed to a string-typed variable is refused (#433).
|
|
50
|
+
`templates/ltx2/keyframes` takes `first_image`/`last_image` as plain
|
|
51
|
+
strings, not `{"location": ...}` (#431/#433).
|
|
52
|
+
- `loop_frames` returns float32 frames in [0, 1] instead of uint8, the shape
|
|
53
|
+
`LTX2ReferenceCondition` needs; a keyframe condition still wants
|
|
54
|
+
`frames_as_array`. `ltx2/reference-sheet`'s default asset is now
|
|
55
|
+
`asset:reference_sheet.jpg` (#444).
|
|
56
|
+
- `validate_workflow` refuses a `components` name the pipeline doesn't
|
|
57
|
+
register; `duration_head` is gone from the in-context LTX-2 templates
|
|
58
|
+
(#442).
|
|
59
|
+
- A `{"media_type": "image"}` reference on a video argument loads as a
|
|
60
|
+
one-frame still (#443).
|
|
61
|
+
- `pair_audio fit: "video"` always fits, and warns on any nonzero gap
|
|
62
|
+
(#428/#429). `concat_videos` and `dissolve_videos` pad a short joined
|
|
63
|
+
track to the frame grid, warning (`joined_audio_padded_to_frames`) only
|
|
64
|
+
when the pad is a frame or more; a residual the AAC mux trims off is
|
|
65
|
+
logged, or warned as `joined_audio_short_after_mux` from a frame up.
|
|
66
|
+
`media.shots` is measured against the file as written (#426/#435/#454).
|
|
67
|
+
Neither warns about resampling inputs that agree to a pinned
|
|
68
|
+
`sample_rate` (#453).
|
|
69
|
+
- New warnings: `match_levels_near_silent` (#434), and `shot_span_overrun`
|
|
70
|
+
from the probes plus a validate-time check (#425).
|
|
71
|
+
- Error text changed: `delete_workspace` (#437/#438), the sub-workflow path
|
|
72
|
+
refusal names the places it looked (#422), and `/outputs/asset:...` misses
|
|
73
|
+
name the asset without server paths.
|
|
74
|
+
|
|
75
|
+
**New**
|
|
76
|
+
|
|
77
|
+
- The `assess_output` tool and `GET /api/gallery/{name}/assess`, plus the
|
|
78
|
+
probe tasks `analyze_shots`, `analyze_seams` and `analyze_sync_drift`
|
|
79
|
+
(#387/#388).
|
|
80
|
+
- A joined video records its shot boundaries (`media.shots`).
|
|
81
|
+
`get_output_frames(seams=true)` uses them, so it no longer needs
|
|
82
|
+
`boundaries` (#385).
|
|
83
|
+
- Run versions (`v<N>`):
|
|
84
|
+
- `list_gallery` returns `run_id`/`version` and filters by `folder` and
|
|
85
|
+
`version`;
|
|
86
|
+
- `output:<wf>/v<N>/<file>` references;
|
|
87
|
+
- `wait_for_job` returns `run_version`;
|
|
88
|
+
- export zips download as `<wf>-vN-<job>.zip`.
|
|
89
|
+
- `list_gallery(media=true)` adds durations, and `output:` names work in
|
|
90
|
+
gallery reads (#356).
|
|
91
|
+
- `DW_PUBLIC_URL` adds absolute URLs to gallery and export responses.
|
|
92
|
+
`export_job` also returns `auth_required` and `open_url` (#353).
|
|
93
|
+
- A `grade` task for images and video: exposure, contrast, saturation and
|
|
94
|
+
temperature/tint (#349).
|
|
95
|
+
- The `templates/minimax/shots-batch` H3 template (#352).
|
|
96
|
+
- Every generative template takes a `seed` argument (#351).
|
|
97
|
+
- `normalize_audio(target_lufs)`, and `integrated_lufs` plus true peak in
|
|
98
|
+
media metadata (#361).
|
|
99
|
+
- `gain_audio` with no region gains the whole track (#395).
|
|
100
|
+
- `world_fade_out_ms` on `assemble-and-score` (#339).
|
|
101
|
+
- Download progress shows in `phase_detail` (#343). `phase_stall` events
|
|
102
|
+
now read as informational (#357).
|
|
103
|
+
- `workflow`, `inline_workflow` and `prompt` also accept a JSON string. A
|
|
104
|
+
mistyped workflow name gets suggestions from the catalog (#397).
|
|
105
|
+
- Host caches are released when each job ends (#368), and the skills point
|
|
106
|
+
at `clear_memory`.
|
|
107
|
+
- `get_job_events(kinds=...)` and `?kinds=` on the event-log route; a kind
|
|
108
|
+
matches an event's `event` or its `kind`, so `["phase_stall"]` selects
|
|
109
|
+
one warning type (#436).
|
|
110
|
+
- `get_memory` reports the step cache's `entries` and `retained_bytes`
|
|
111
|
+
(#418).
|
|
112
|
+
- `get_output_image` and `/outputs` resolve `asset:` references (#445), and
|
|
113
|
+
`get_output_frames(seams=true)` works on linked assets (#430).
|
|
114
|
+
- Compact `assess_output` lists each finding once (#427). Shots are named by
|
|
115
|
+
their source when joined inputs already carry shots (#432).
|
|
116
|
+
- A task-only workflow's run history counts, so its estimate can quote
|
|
117
|
+
`basis: observed` (#439). The Music 3 hint no longer shows on video
|
|
118
|
+
(#441).
|
|
119
|
+
|
|
120
|
+
**Fixes**
|
|
121
|
+
|
|
122
|
+
- The step cache's retained-byte count no longer only grows (#418).
|
|
123
|
+
- `templates/ltx2/keyframes` (#431), `restore-decompression` (#442) and
|
|
124
|
+
`reference-sheet` (#444) run with their own defaults again.
|
|
125
|
+
- Joined audio and shot maps stay on the frame grid through repeated joins
|
|
126
|
+
(#423, #426, #428, #435).
|
|
127
|
+
|
|
128
|
+
Releases are cut by pushing a `v<semver>` tag. CI does the rest.
|
|
129
|
+
|
|
130
|
+
Before merging `develop` into `master`, run `scripts/preflight.sh` and get it
|
|
131
|
+
passing. It covers more than CI: ruff over the whole repo rather than
|
|
132
|
+
`dw dw_mcp tests`, and the UI's Playwright e2e tests, which CI doesn't run.
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
scripts/release.sh 0.38.0
|
|
136
|
+
scripts/release.sh 0.38.0-alpha.1 "UI front end" # optional tag message
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
The script bumps `pyproject.toml` (the single source of the version —
|
|
140
|
+
`dw.__version__` reads it at runtime) and sets the same version in
|
|
141
|
+
`plugins/dw/.claude-plugin/plugin.json`, so an installed plugin names the
|
|
142
|
+
engine it was written against; it commits just those two files, pushes
|
|
143
|
+
master, tags the bump commit `v0.38.0`, and pushes the tag. It refuses
|
|
144
|
+
a malformed version, a branch other than master, an existing tag, or a
|
|
145
|
+
dirty index (unstaged changes elsewhere are fine — the release commit
|
|
146
|
+
is path-limited to those two files).
|
|
147
|
+
|
|
148
|
+
By hand, the equivalent is:
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
# 1. Bump the version in pyproject.toml:
|
|
152
|
+
# version = "0.38.0"
|
|
153
|
+
# 2. Set the same version in plugins/dw/.claude-plugin/plugin.json
|
|
154
|
+
git commit -m "release 0.38.0" -- pyproject.toml plugins/dw/.claude-plugin/plugin.json
|
|
155
|
+
|
|
156
|
+
# 3. Tag the bump commit and push
|
|
157
|
+
git tag -a v0.38.0 -m "release 0.38.0"
|
|
158
|
+
git push origin master v0.38.0
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
The tag must point at a commit whose pyproject already declares the
|
|
162
|
+
same version — the release job checks and refuses a mismatch.
|
|
163
|
+
|
|
164
|
+
The tag triggers the full CI chain: backend tests, UI lint/type-check/
|
|
165
|
+
unit tests, then the wheel build (SPA compiled into the package via
|
|
166
|
+
`scripts/build_dist.sh`). Only if all of that passes does the `release`
|
|
167
|
+
job run — it verifies the tag matches the pyproject version, then
|
|
168
|
+
creates a GitHub release named after the tag with auto-generated notes
|
|
169
|
+
and the wheel + sdist attached.
|
|
170
|
+
|
|
171
|
+
Note on pre-release numbering: Python packaging normalizes semver-style
|
|
172
|
+
pre-releases, so a `0.38.0-alpha.1` version builds a wheel named
|
|
173
|
+
`0.38.0a1`. The tag, pyproject, and release stay in the semver form;
|
|
174
|
+
only the wheel filename and pip metadata show the normalized one.
|
|
175
|
+
|
|
176
|
+
A pre-release tag like `v0.38.0-rc1` is marked as a pre-release on
|
|
177
|
+
GitHub. Tags that aren't `v` + semver (or that don't match the declared
|
|
178
|
+
versions) fail the release job before anything is published.
|
|
179
|
+
|
|
180
|
+
After the GitHub release, the `pypi` job publishes the same artifacts to
|
|
181
|
+
PyPI via [trusted publishing](https://docs.pypi.org/trusted-publishers/)
|
|
182
|
+
(OIDC — no token stored anywhere). One-time setup on pypi.org under
|
|
183
|
+
*Publishing*: add a trusted publisher for project `diffusers-workflow`
|
|
184
|
+
with owner `dkackman`, repository `diffusers-workflow`, workflow
|
|
185
|
+
`ci.yml`, environment `pypi` (use "add a pending publisher" before the
|
|
186
|
+
first release, since the project won't exist yet). Pre-release versions
|
|
187
|
+
are hidden from plain `pip install`; they need `pip install --pre`.
|
|
188
|
+
|
|
189
|
+
Note: released `diffusers` from PyPI may lag the newest model pipelines
|
|
190
|
+
this project targets — a PyPI install can need
|
|
191
|
+
`pip install git+https://github.com/huggingface/diffusers` on top.
|
|
192
|
+
|
|
193
|
+
To rebuild artifacts without releasing, run the CI workflow manually
|
|
194
|
+
(`workflow_dispatch`) — the wheel job uploads `dist/*` as a workflow
|
|
195
|
+
artifact.
|