diffusers-workflow 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
  2. diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
  3. diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
  4. diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
  5. diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
  6. diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
  7. dw/__init__.py +440 -0
  8. dw/adapter_compatibility.py +226 -0
  9. dw/arguments.py +1231 -0
  10. dw/assessment_rules.py +159 -0
  11. dw/assets.py +130 -0
  12. dw/cache_blocks.json +16 -0
  13. dw/cache_blocks.py +146 -0
  14. dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
  15. dw/content_types.py +150 -0
  16. dw/dissolve_frame_errors.py +121 -0
  17. dw/docs/ACCELERATION.md +352 -0
  18. dw/docs/AGENT_LOOP.md +95 -0
  19. dw/docs/DEPENDENCIES.md +91 -0
  20. dw/docs/IP_ADAPTER.md +109 -0
  21. dw/docs/LORAS.md +131 -0
  22. dw/docs/MCP.md +517 -0
  23. dw/docs/PROMPT_WEIGHTING.md +78 -0
  24. dw/docs/QUANTIZATION.md +230 -0
  25. dw/docs/RECIPES_24GB.md +201 -0
  26. dw/docs/RELEASING.md +195 -0
  27. dw/docs/REMOTE.md +140 -0
  28. dw/docs/REPL_COMMANDS.md +121 -0
  29. dw/docs/REPL_WORKER_GUIDE.md +51 -0
  30. dw/docs/SECURITY.md +272 -0
  31. dw/docs/SECURITY_QUICKREF.md +112 -0
  32. dw/docs/SERVER.md +679 -0
  33. dw/docs/TASKS.md +1741 -0
  34. dw/docs/TESTING.md +71 -0
  35. dw/docs/WORKFLOW_GUIDE.md +2038 -0
  36. dw/docs/WORKSPACES.md +316 -0
  37. dw/download_watch.py +335 -0
  38. dw/elision.py +306 -0
  39. dw/events.py +275 -0
  40. dw/for_each.py +409 -0
  41. dw/host_memory.py +258 -0
  42. dw/host_memory_projection.py +230 -0
  43. dw/hub_cache.py +432 -0
  44. dw/introspection.py +1228 -0
  45. dw/kernel_availability.py +208 -0
  46. dw/locations.py +599 -0
  47. dw/log_setup.py +45 -0
  48. dw/loudness.py +82 -0
  49. dw/media_audio.py +217 -0
  50. dw/media_frames.py +367 -0
  51. dw/media_info.py +297 -0
  52. dw/pipeline_processors/chain.py +821 -0
  53. dw/pipeline_processors/config_objects.py +237 -0
  54. dw/pipeline_processors/pipeline.py +2297 -0
  55. dw/pipeline_processors/remote.py +46 -0
  56. dw/plan.py +920 -0
  57. dw/previous_results.py +411 -0
  58. dw/probe_paths.py +59 -0
  59. dw/prompt_schema.json +48 -0
  60. dw/prompt_weighting.py +378 -0
  61. dw/prompts.py +159 -0
  62. dw/realize.py +250 -0
  63. dw/reference_limits.py +215 -0
  64. dw/reference_names.py +125 -0
  65. dw/repl.py +338 -0
  66. dw/repl_commands.py +836 -0
  67. dw/repl_worker.py +159 -0
  68. dw/result.py +1720 -0
  69. dw/result_fps.py +82 -0
  70. dw/run.py +162 -0
  71. dw/runs.py +768 -0
  72. dw/scalar_result_validation.py +97 -0
  73. dw/schema.py +283 -0
  74. dw/security.py +1038 -0
  75. dw/select_validation.py +115 -0
  76. dw/serve.py +277 -0
  77. dw/server/__init__.py +2 -0
  78. dw/server/app.py +4586 -0
  79. dw/server/assess.py +132 -0
  80. dw/server/catalog_shape.py +487 -0
  81. dw/server/enhancers.py +129 -0
  82. dw/server/exports.py +480 -0
  83. dw/server/guides.py +257 -0
  84. dw/server/jobs.py +1561 -0
  85. dw/server/mcp_mount.py +95 -0
  86. dw/server/netinfo.py +124 -0
  87. dw/server/observed_cost.py +379 -0
  88. dw/server/sysinfo.py +71 -0
  89. dw/server/ui/assets/abap-08VXUWAP.js +1 -0
  90. dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
  91. dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
  92. dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
  93. dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
  94. dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
  95. dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
  96. dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
  97. dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
  98. dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
  99. dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
  100. dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
  101. dw/server/ui/assets/css-DIMkf-bt.js +3 -0
  102. dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
  103. dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
  104. dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
  105. dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
  106. dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
  107. dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
  108. dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
  109. dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
  110. dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
  111. dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
  112. dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
  113. dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
  114. dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
  115. dw/server/ui/assets/go-C-y9NEjX.js +1 -0
  116. dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
  117. dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
  118. dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
  119. dw/server/ui/assets/html-PhsdjHSr.js +1 -0
  120. dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
  121. dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
  122. dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
  123. dw/server/ui/assets/index-DgrYhQd9.js +43 -0
  124. dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
  125. dw/server/ui/assets/java-BEtHBSE6.js +1 -0
  126. dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
  127. dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
  128. dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
  129. dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
  130. dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
  131. dw/server/ui/assets/less-B9JPFI3C.js +2 -0
  132. dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
  133. dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
  134. dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
  135. dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
  136. dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
  137. dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
  138. dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
  139. dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
  140. dw/server/ui/assets/monaco--ixms01u.css +1 -0
  141. dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
  142. dw/server/ui/assets/msdax-DauUninz.js +1 -0
  143. dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
  144. dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
  145. dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
  146. dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
  147. dw/server/ui/assets/perl-oz_6vUea.js +1 -0
  148. dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
  149. dw/server/ui/assets/php-nr791fC2.js +1 -0
  150. dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
  151. dw/server/ui/assets/postiats-43DmfD33.js +1 -0
  152. dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
  153. dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
  154. dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
  155. dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
  156. dw/server/ui/assets/python-Bcn70HdC.js +1 -0
  157. dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
  158. dw/server/ui/assets/r-BwWrilGY.js +1 -0
  159. dw/server/ui/assets/razor-D1HmNnby.js +1 -0
  160. dw/server/ui/assets/redis-ClamHrr6.js +1 -0
  161. dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
  162. dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
  163. dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
  164. dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
  165. dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
  166. dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
  167. dw/server/ui/assets/scheme-BeGwcela.js +1 -0
  168. dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
  169. dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
  170. dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
  171. dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
  172. dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
  173. dw/server/ui/assets/sql-NEE52Syq.js +1 -0
  174. dw/server/ui/assets/st-DbInun42.js +1 -0
  175. dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
  176. dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
  177. dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
  178. dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
  179. dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
  180. dw/server/ui/assets/twig-De2hgUGE.js +1 -0
  181. dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
  182. dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
  183. dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
  184. dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
  185. dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
  186. dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
  187. dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
  188. dw/server/ui/index.html +17 -0
  189. dw/server/updater.py +192 -0
  190. dw/settings.py +98 -0
  191. dw/shot_span_preflight.py +116 -0
  192. dw/shots.py +359 -0
  193. dw/slice_preflight.py +148 -0
  194. dw/step.py +187 -0
  195. dw/step_cache.py +442 -0
  196. dw/subfolders.py +107 -0
  197. dw/task_domains.py +307 -0
  198. dw/tasks/assess.py +826 -0
  199. dw/tasks/audio_transcription.py +88 -0
  200. dw/tasks/audio_utils.py +1862 -0
  201. dw/tasks/background_remover.py +43 -0
  202. dw/tasks/borders.py +113 -0
  203. dw/tasks/compose_text.py +74 -0
  204. dw/tasks/concat_videos.py +300 -0
  205. dw/tasks/depth_estimator.py +54 -0
  206. dw/tasks/diffusion_upscale.py +109 -0
  207. dw/tasks/dissolve_videos.py +342 -0
  208. dw/tasks/format_messages.py +24 -0
  209. dw/tasks/gather.py +173 -0
  210. dw/tasks/grade.py +97 -0
  211. dw/tasks/image_to_text.py +43 -0
  212. dw/tasks/image_utils.py +764 -0
  213. dw/tasks/interpolate_frames.py +252 -0
  214. dw/tasks/judge.py +68 -0
  215. dw/tasks/model_cache.py +55 -0
  216. dw/tasks/pair_audio.py +268 -0
  217. dw/tasks/qr_code.py +19 -0
  218. dw/tasks/restore_faces.py +175 -0
  219. dw/tasks/rife_model.py +192 -0
  220. dw/tasks/segment.py +121 -0
  221. dw/tasks/select.py +111 -0
  222. dw/tasks/speech_generation.py +228 -0
  223. dw/tasks/stabilize.py +129 -0
  224. dw/tasks/task.py +920 -0
  225. dw/tasks/tensor_image.py +57 -0
  226. dw/tasks/text_generation.py +169 -0
  227. dw/tasks/text_sections.py +80 -0
  228. dw/tasks/upscale.py +203 -0
  229. dw/tasks/video_utils.py +624 -0
  230. dw/tasks/zoe_depth.py +71 -0
  231. dw/teacache.py +381 -0
  232. dw/teacache_models.json +99 -0
  233. dw/test.py +29 -0
  234. dw/type_helpers.py +231 -0
  235. dw/validate.py +68 -0
  236. dw/variable_constraints.py +444 -0
  237. dw/variables.py +443 -0
  238. dw/video_extensions.py +141 -0
  239. dw/vram_estimate.py +116 -0
  240. dw/worker.py +764 -0
  241. dw/workflow.py +2007 -0
  242. dw/workflow_schema.json +1346 -0
  243. dw/workflow_sources.py +383 -0
  244. dw/workflows/h3_context_ir.json +57 -0
  245. dw/workflows/test.json +31 -0
  246. dw/workspace.py +730 -0
  247. dw_mcp/__init__.py +6 -0
  248. dw_mcp/__main__.py +133 -0
  249. dw_mcp/assets.py +336 -0
  250. dw_mcp/authoring.py +114 -0
  251. dw_mcp/catalog.py +360 -0
  252. dw_mcp/client.py +486 -0
  253. dw_mcp/diagnose.py +371 -0
  254. dw_mcp/exports.py +84 -0
  255. dw_mcp/guides.py +35 -0
  256. dw_mcp/media.py +638 -0
  257. dw_mcp/models.py +97 -0
  258. dw_mcp/prompts.py +104 -0
  259. dw_mcp/server.py +1343 -0
  260. dw_mcp/workspaces.py +212 -0
@@ -0,0 +1,230 @@
1
+ # Quantization
2
+
3
+ Quantization reduces model memory usage by storing weights at lower precision. diffusers-workflow ships examples for BitsAndBytes, TorchAO, GGUF, and SDNQ, applied per-component in the pipeline - and any other backend with a config class (optimum-quanto, for example) works through the same dynamic `config_type` import.
4
+
5
+ ## Per-Component Quantization
6
+
7
+ Quantize individual components (transformer, text encoder, etc.) independently:
8
+
9
+ ```json
10
+ {
11
+ "pipeline": {
12
+ "transformer": {
13
+ "configuration": { "component_type": "FluxTransformer2DModel" },
14
+ "quantization_config": {
15
+ "configuration": { "config_type": "..." },
16
+ "arguments": { ... }
17
+ },
18
+ "from_pretrained_arguments": {
19
+ "model_name": "...",
20
+ "subfolder": "transformer",
21
+ "torch_dtype": "torch.bfloat16"
22
+ }
23
+ },
24
+ "configuration": { "component_type": "FluxPipeline" },
25
+ "from_pretrained_arguments": {
26
+ "model_name": "...",
27
+ "torch_dtype": "torch.bfloat16"
28
+ }
29
+ }
30
+ }
31
+ ```
32
+
33
+ The component is loaded separately with quantization, then the rest of the pipeline loads around it.
34
+
35
+ ## BitsAndBytes (CUDA only)
36
+
37
+ 4-bit and 8-bit quantization via bitsandbytes:
38
+
39
+ ```json
40
+ "quantization_config": {
41
+ "configuration": { "config_type": "BitsAndBytesConfig" },
42
+ "arguments": {
43
+ "load_in_4bit": true,
44
+ "bnb_4bit_quant_type": "{nf4}",
45
+ "bnb_4bit_compute_dtype": "torch.bfloat16"
46
+ }
47
+ }
48
+ ```
49
+
50
+ Note: `"{nf4}"` uses braces to keep the string literal. Without braces, the type system would try to load `nf4` as a Python class.
51
+
52
+ For 8-bit:
53
+
54
+ ```json
55
+ "arguments": { "load_in_8bit": true }
56
+ ```
57
+
58
+ **Example:** [flux2-dev.json](../workflows/models/flux2-dev.json) (a pre-quantized 4-bit checkpoint)
59
+
60
+ ## TorchAO
61
+
62
+ Quantization via TorchAO. `quant_type` must be an `AOBaseConfig` class (diffusers no
63
+ longer accepts string shorthands like `"int4wo"`). Name the class as a dotted
64
+ `quant_type` and it is instantiated automatically with no arguments before being passed
65
+ to `TorchAoConfig`:
66
+
67
+ ```json
68
+ "quantization_config": {
69
+ "configuration": { "config_type": "TorchAoConfig" },
70
+ "arguments": {
71
+ "quant_type": "torchao.quantization.Int8WeightOnlyConfig",
72
+ "modules_to_not_convert": ["proj_in", "proj_out"]
73
+ }
74
+ }
75
+ ```
76
+
77
+ Common choices: `Int8WeightOnlyConfig` (any CUDA card), `Int4WeightOnlyConfig` (smallest),
78
+ `Float8DynamicActivationFloat8WeightConfig` (fastest, requires compute capability 8.9+ -
79
+ RTX 40-series/Ada or newer).
80
+
81
+ **Example:** [flux-torchao.json](../workflows/models/flux-torchao.json)
82
+
83
+ **Pair TorchAO with `torch.compile`.** Int8 weight-only and float8 dynamic-activation
84
+ quant types get their fused-kernel speedups only under compilation - uncompiled they are
85
+ a memory win but often a speed *loss*. Compiled is not a guarantee either: int8 weight-only
86
+ on an RTX 3090 measured over a minute per denoising step for Flux dev, compiled, against
87
+ two seconds for bf16 (see [RECIPES_24GB.md](RECIPES_24GB.md#flux-dev-12b)). Measure on the
88
+ card before recording a TorchAO recipe as fast. Add a `compile` block to the quantized component
89
+ (see [ACCELERATION.md](ACCELERATION.md#torchcompile)):
90
+
91
+ ```json
92
+ "configuration": {
93
+ "components": {
94
+ "transformer": {
95
+ "compile": { "repeated_blocks": true, "fullgraph": true }
96
+ }
97
+ }
98
+ }
99
+ ```
100
+
101
+ ## GGUF
102
+
103
+ Load GGUF-format checkpoint files:
104
+
105
+ ```json
106
+ "quantization_config": {
107
+ "configuration": { "config_type": "GGUFQuantizationConfig" },
108
+ "arguments": {
109
+ "compute_dtype": "torch.bfloat16"
110
+ }
111
+ }
112
+ ```
113
+
114
+ GGUF models load from single files using `from_single_file`:
115
+
116
+ ```json
117
+ "from_pretrained_arguments": {
118
+ "from_single_file": "https://huggingface.co/city96/FLUX.1-dev-gguf/blob/main/flux1-dev-Q2_K.gguf",
119
+ "torch_dtype": "torch.bfloat16"
120
+ }
121
+ ```
122
+
123
+ **Example:** [flux-gguf.json](../workflows/models/flux-gguf.json)
124
+
125
+ ## SDNQ (SD.Next Quantization)
126
+
127
+ SDNQ works two ways: quantize a component on the fly at load time, or load a
128
+ pre-quantized model as a complete pipeline.
129
+
130
+ ### On-the-fly (`sdnq.SDNQConfig`)
131
+
132
+ Quantizes the component while it loads - the pattern the LTX-2 and MiniMax H3
133
+ examples use for their large transformers and text encoders:
134
+
135
+ ```json
136
+ "quantization_config": {
137
+ "configuration": { "config_type": "sdnq.SDNQConfig" },
138
+ "arguments": {
139
+ "weights_dtype": "{uint4}",
140
+ "quantization_device": "cuda",
141
+ "return_device": "cuda",
142
+ "use_quantized_matmul": true,
143
+ "dequantize_fp32": false
144
+ }
145
+ }
146
+ ```
147
+
148
+ - `weights_dtype` — the storage dtype (`uint4`, `int8`, ...; brace-escaped so it stays a string)
149
+ - `quantization_device` / `return_device` — where the quantization pass runs and where the finished component lands; quantizing on `cuda` is much faster than on CPU
150
+ - `use_quantized_matmul` — quantized matmul kernels (CUDA/XPU only)
151
+
152
+ **Examples:** [text-to-video.json](../workflows/templates/ltx2/text-to-video.json), [video-with-audio.json](../workflows/templates/minimax/video-with-audio.json)
153
+
154
+ ### Pre-quantized models
155
+
156
+ Pre-quantized SDNQ repos load as complete pipelines. The `sdnq` module must be imported before loading so it can register with diffusers:
157
+
158
+ ```json
159
+ {
160
+ "pipeline": {
161
+ "configuration": {
162
+ "component_type": "ZImagePipeline",
163
+ "pre_load_modules": ["sdnq"],
164
+ "sdnq_optimize": ["transformer", "text_encoder"]
165
+ },
166
+ "from_pretrained_arguments": {
167
+ "model_name": "Disty0/Z-Image-Turbo-SDNQ-uint4-svd-r32",
168
+ "torch_dtype": "torch.bfloat16"
169
+ }
170
+ }
171
+ }
172
+ ```
173
+
174
+ - `pre_load_modules` — Imports sdnq before pipeline loading (registers quantization method)
175
+ - `sdnq_optimize` — Applies quantized matmul to listed components (CUDA/XPU only, skipped on MPS/CPU)
176
+
177
+ **Example:** [z-image-sdnq.json](../workflows/models/z-image-sdnq.json)
178
+
179
+ ## Modular Pipelines
180
+
181
+ A modular pipeline pulls its component weights itself via `load_components()` rather than
182
+ through `from_pretrained_arguments`, so quantization is keyed by component name under
183
+ `load_components.quantization_config` instead of living on a separate component block:
184
+
185
+ ```json
186
+ "configuration": {
187
+ "component_type": "MiniMaxMusic3ModularPipeline",
188
+ "load_components": {
189
+ "dtype": "torch.bfloat16",
190
+ "quantization_config": {
191
+ "transformer": {
192
+ "configuration": { "config_type": "TorchAoConfig" },
193
+ "arguments": { "quant_type": "torchao.quantization.Int8WeightOnlyConfig" }
194
+ },
195
+ "language_model": {
196
+ "configuration": { "config_type": "transformers.TorchAoConfig" },
197
+ "arguments": { "quant_type": "torchao.quantization.Int8WeightOnlyConfig" }
198
+ }
199
+ }
200
+ }
201
+ }
202
+ ```
203
+
204
+ A component the map does not name loads unquantized. Note that a transformers-based
205
+ component (a text encoder, for example) takes the `transformers.TorchAoConfig` class,
206
+ not the diffusers one - the `config_type` still resolves either via the dynamic import
207
+ described below.
208
+
209
+ ## Custom Quantization
210
+
211
+ Any quantization backend that provides a config class works via the `config_type` field with a dotted module path:
212
+
213
+ ```json
214
+ "quantization_config": {
215
+ "configuration": { "config_type": "some_package.SomeQuantConfig" },
216
+ "arguments": { ... }
217
+ }
218
+ ```
219
+
220
+ The class is loaded dynamically via importlib.
221
+
222
+ ## Platform Notes
223
+
224
+ | Framework | CUDA | MPS | CPU |
225
+ | --------- | ---- | --- | --- |
226
+ | BitsAndBytes | Yes | No | No |
227
+ | TorchAO | Yes | Partial | No |
228
+ | GGUF | Yes | Yes | Yes |
229
+ | SDNQ (load) | Yes | Yes | Yes |
230
+ | SDNQ (optimize) | Yes | No | No |
@@ -0,0 +1,201 @@
1
+ # Fast on 24GB
2
+
3
+ Recommended configurations per model family for a 24GB consumer GPU (RTX 3090/4090 class). Every knob here is documented individually in [ACCELERATION.md](ACCELERATION.md) and [QUANTIZATION.md](QUANTIZATION.md); this page is about the combinations that work.
4
+
5
+ The general recipe, in order of impact:
6
+
7
+ 1. **Fit the transformer first.** If it fits in bf16 with room for activations, don't quantize. If it doesn't, prefer float8/int8 quantization (TorchAO, GGUF Q8) over offloading - quantization costs quality once, offloading costs speed every step.
8
+ 2. **Compile the transformer** (`"compile": {"repeated_blocks": true}`). 1.3-1.5x, stacks with everything below. The REPL worker keeps compiled pipelines loaded, so the compile cost is paid once per session. Add `fullgraph: true` only when no cache is configured - cache hooks need a graph break.
9
+ 3. **Cache** (`"cache": {"type": "first_block"}`). Another 1.5-2x at mild quality cost; raise `threshold` to taste.
10
+ 4. **Offload only what doesn't fit.** Text encoders and VAE tolerate `offload: "model"` cheaply - they run once per generation, not once per step. In a modular pipeline the same components take `"residency": "on_demand"`, which frees their VRAM for the denoise loop at the cost of one pair of transfers per call.
11
+ 5. **Pin the attention backend** on compiled components (`"attention_backend": "flash_hub"` or `"sage_hub"` - fetched from the Hub, no local build).
12
+
13
+ ## Flux dev (12B)
14
+
15
+ The bf16 transformer is ~24GB - it does not fit alongside the T5 encoder, and it does
16
+ not fit alongside the VAE either: a resident bf16 transformer runs the denoise and then
17
+ fails at decode. `offload: "model"` is what makes bf16 work on 24GB, and on an RTX 3090
18
+ it is also the fastest configuration measured (1024x1024, 28 steps, pipeline loaded):
19
+
20
+ | Approach | Config | Measured on RTX 3090 |
21
+ | -------- | ------ | -------------------- |
22
+ | **bf16 + model offload** | [flux-dev.json](../workflows/models/flux-dev.json) | 55s per image (72s cold) |
23
+ | **bf16 + model offload + compile** | [flux-dev-compile.json](../workflows/models/flux-dev-compile.json) - `compile: {repeated_blocks: true}` on the transformer | 52s per image (64s cold) |
24
+ | **int8 TorchAO** | transformer `quant_type: "torchao.quantization.Int8WeightOnlyConfig"` + `compile`, with or without `cache: first_block` | over 60s per *denoising step* - do not use on Ampere |
25
+ | **float8 TorchAO** (RTX 40-series+) | `Float8DynamicActivationFloat8WeightConfig` + `compile` | needs compute capability 8.9+ (Ada); unmeasured |
26
+ | **GGUF Q8** | [flux-gguf.json](../workflows/models/flux-gguf.json) - `from_single_file` Q8_0 transformer + `offload: "model"` | unmeasured; smallest VRAM, best quality retention |
27
+
28
+ Measured 2026-09-07. A number in this table came from a run on the named card; a row
29
+ without one is a configuration that loads, not a speed claim.
30
+
31
+ **Examples:** [flux-dev-compile.json](../workflows/models/flux-dev-compile.json), [flux-gguf.json](../workflows/models/flux-gguf.json), [step-caching.json](../workflows/templates/step-caching.json)
32
+
33
+ ## Qwen-Image (20B)
34
+
35
+ Too large for bf16 on 24GB. Quantize the transformer (TorchAO int8/float8 or GGUF Q4/Q5) and model-offload the rest; add `compile` + `first_block` cache as for Flux. The Qwen2.5-VL text encoder is large - quantize it too, or group-offload it (`"text_encoder.model"` with `leaf_level`).
36
+
37
+ The catalog no longer ships a Qwen-Image workflow - author one from [text-to-image.json](../workflows/templates/text-to-image.json) with the configuration above.
38
+
39
+ ## Wan 2.2
40
+
41
+ - **TI2V-5B**: fits in bf16 on 24GB. No quantization needed - just `compile` + `cache` + VAE tiling for longer clips.
42
+ - **T2V/I2V-A14B**: two 14B transformers (`transformer` + `transformer_2`). Quantize both (GGUF Q4/Q5 or TorchAO int8) and use `offload: "model"`; enable `vae.enable_tiling`.
43
+
44
+ The catalog no longer ships Wan workflows - author one from [minimax/image-to-video.json](../workflows/templates/minimax/image-to-video.json) with the configuration above.
45
+
46
+ ## HunyuanVideo (13B)
47
+
48
+ Same shape as Flux: quantize the transformer (GGUF Q6/Q8, or TorchAO int8/float8 per the Flux table) + `offload: "model"` + `vae.enable_tiling`. The Llama text encoder benefits from group offload. Use `first_block` cache - video steps are expensive, caching pays off more than on images.
49
+
50
+ The catalog no longer ships a HunyuanVideo workflow - author one from [ltx2/text-to-video.json](../workflows/templates/ltx2/text-to-video.json) with the configuration above.
51
+
52
+ ## MiniMax-H3
53
+
54
+ A modular pipeline, so everything is per component in `components` rather than a
55
+ pipeline-level `offload`. The working 24GB configuration at 960x544:
56
+
57
+ | Component | Config |
58
+ | --------- | ------ |
59
+ | `transformer` / `transformer_ref` | SDNQ int4 (`quantization_device: "cuda"`, `return_device: "cpu"`), `group_offload` `block_level` with `num_blocks_per_group: 1-2` and `use_stream: true` |
60
+ | `text_encoder` | `remove_modules: ["lm_head"]` - the encoder path never calls it |
61
+ | `text_encoder.model` | SDNQ int4, `truncate_layers: {"language_model.layers": 51}`, `group_offload` `leaf_level` |
62
+ | `vae`, `audio_vae` | SDNQ int8, `device: "cuda"`, `residency: "on_demand"` |
63
+ | pipeline | `cache: first_block` (`threshold: 0.1`) on the 20-step ref2va workflows only - measured on the 9-step turbo schedule it never skips (consecutive distilled steps differ too much for the threshold), so the turbo examples omit it rather than hold cache state for nothing |
64
+
65
+ The VAEs are the piece worth calling out. They hold roughly 3GiB, are used only to
66
+ encode references and decode the result, and group offloading them is worse than useless
67
+ because tiled decode restreams the model once per tile. Every H3 workflow adds
68
+ `"residency": "on_demand"` to both. The reference workflows, which carry the most
69
+ conditioning, go from 23.2GiB peak reserved with 40 allocator retries to 18.9GiB with
70
+ none; the frame-conditioned ones from 22.7GiB with 22 retries to 18.0GiB with none. Both
71
+ for about 1% in wall time.
72
+ Spend the headroom on length: carrying a frame between chained segments adds a reference
73
+ and ~1.9GiB, which is what made the chained variants OOM on their second segment before.
74
+
75
+ The text encoder pruning exists because H3 conditions on `hidden_states[50]` of its
76
+ 64-layer Qwen3-VL: layers 51-63 and the LM head run (and stream) for nothing on every
77
+ encode. Keeping 51 layers is bit-identical - index 50 of the tuple is recorded before
78
+ layer 50 runs; keeping only 50 would hand back the final-norm output, a different
79
+ tensor - and it returns a few GiB of system RAM on a host that needs every one of them
80
+ (a full t2va run peaks around 63GiB RSS on a 64GiB box). Note H3's VAE constructs with
81
+ tiling already enabled, so a pipeline-level `vae.enable_tiling` adds nothing here.
82
+
83
+ Two levers measured and *rejected* on a 3090 (A/B, t2va 960x544x124f, 2026-08):
84
+ `low_cpu_mem_usage: false` on the transformer's group offload (pin host copies once
85
+ instead of re-pinning per onload) left step time unchanged at ~15.4s - at int4 the
86
+ step is compute-bound and the transfers already hide under it - while the ~27GiB of
87
+ unswappable pinned memory pushed the host into an OOM kill. `use_stream` on the text
88
+ encoder's leaf offload was backed out with it for the same host-memory reason. On a
89
+ faster GPU (or a host with more RAM) both are worth re-testing; the step budget there
90
+ may actually expose the transfer time.
91
+
92
+ Length costs VRAM but the configuration holds to the model's full range: a single
93
+ 345-frame take (14.4s, the `17n+5` maximum) peaks at 23.6GiB reserved at 960x544 -
94
+ inside 24GB with nothing to spare - and denoises in ~13 minutes on a 3090 with the
95
+ 9-step turbo schedule (~85-100s a step once warm, against ~15s at 124 frames).
96
+
97
+ Host RAM is the tighter budget than VRAM on a 64GiB box. Loading H3 peaks around
98
+ 59GiB RSS and a running ref2va shot sits at 45-53GiB, so a workflow that ran another
99
+ model first (Z-Image drawing a subject, Music3 writing a song) must free it with
100
+ `release_pipeline` before H3 loads - with it, the multi-model digital-short
101
+ workflows below fit; without it, the load is an OOM kill, not a slowdown.
102
+
103
+ **Examples:** [reference-to-video.json](../workflows/templates/minimax/reference-to-video.json), [chain-matched-to-audio.json](../workflows/templates/minimax/chain-matched-to-audio.json), [image-to-video.json](../workflows/templates/minimax/image-to-video.json), [dialogue-short.json](../workflows/templates/minimax/dialogue-short.json) (five ref2va shots + two Z-Image portraits in ~35 minutes end to end)
104
+
105
+ ## MiniMax-Music3
106
+
107
+ Music3 runs at about 22GiB in bfloat16 under the templates' `components_manager`
108
+ auto CPU offload, which keeps only the running component resident; no quantization
109
+ is needed on a 24GB card. The language model is the part worth offloading harder:
110
+ a leaf-level `group_offload` of `language_model` brings it to about 8GiB (the model
111
+ card's low-VRAM recipe). Two things the examples carry: `release_pipeline` on the
112
+ music step in any workflow that loads H3 afterwards, since host RAM is the binding
113
+ constraint (see Multi-model workflows below), and the run's time follows the length
114
+ the model actually sings, not `audio_duration`, which is a ceiling of at most 9000
115
+ frames at 25 frames per second (360 seconds). Output is 44.1 kHz stereo.
116
+
117
+ **Examples:** [music.json](../workflows/templates/minimax/music.json), [music-video.json](../workflows/templates/minimax/music-video.json)
118
+
119
+ ## LTX-2.5 (22B, video + audio)
120
+
121
+ A standard pipeline, but placed per component rather than with a pipeline-level
122
+ `offload` - the transformer is the only thing that wants to be resident, and the text
123
+ encoder is nearly as large as it is. The working 24GB configuration at 960x544:
124
+
125
+ | Component | Config |
126
+ | --------- | ------ |
127
+ | `transformer` | SDNQ `uint4` (`quantization_device: "cuda"`, `return_device: "cuda"`, `use_quantized_matmul: true`), resident |
128
+ | `text_encoder` (Gemma 4, 23GB) | SDNQ `int8`, `return_device: "cpu"`, `group_offload` `leaf_level` with `use_stream: true` |
129
+ | `connectors` (12GB) | `group_offload` `leaf_level` with `use_stream: true` |
130
+ | `vae`, `audio_vae`, `vocoder`, `duration_head` | `device: "cuda"` - small, and used once per generation |
131
+ | pipeline | `vae.enable_tiling` for anything above the base resolution |
132
+
133
+ Two things about the checkpoint are worth knowing before tuning anything:
134
+
135
+ - **`transformer` is the distilled model.** It runs a fixed 8-step schedule at
136
+ `guidance_scale: 1.0`, with STG and modality guidance off, and the `sigmas` every
137
+ example passes are its trained schedule - not a knob. They are referenced from
138
+ diffusers (`constant:diffusers.pipelines.ltx2.utils.DISTILLED_SIGMA_VALUES`) rather
139
+ than copied, so the schedule stays whatever the library says it is. `num_inference_steps`,
140
+ `guidance_scale`, `stg_scale` and the rest only mean anything against
141
+ `subfolder: "transformer_full"`, the dev model, which is not a 24GB configuration:
142
+ it is the same ~38GB in bf16, and the guidance those knobs turn on costs three
143
+ transformer passes per step against CFG-doubled batches. Nothing here ships it.
144
+ - **The checkpoint ships a diffusion decoder that `LTX2Pipeline` ignores**, and on
145
+ 24GB you are not missing much. It is listed in `model_index.json` but is not a
146
+ constructor argument, so diffusers logs "not expected ... will be ignored" and
147
+ decodes with the convolutional VAE. Reaching it means `LTX2VideoDiffusionDecodePipeline`
148
+ on a step run with `output_type: "{latent}"`, and two things get in the way. Its
149
+ neighborhood attention has two processors, and the one you get by default is the
150
+ portable FlexAttention fallback: it densifies a `seq_len x seq_len` block mask and then
151
+ runs uncompiled `flex_attention`, which falls to the eager reference path. Both
152
+ allocations are quadratic in the output grid, and neither is reduced by tiling (stages
153
+ 1-3 always run on the full volume) or by shrinking the clip (the stage-4 grid is near
154
+ output resolution either way). Measured on an otherwise empty 3090 (#153): 10.05GiB
155
+ inside stages 1-3 at 960x544x121, 69.77GiB for one stage-4 attention at 512x288x25
156
+ (17.44GiB just to densify that stage's mask, whichever it reaches first), and ~25.5GiB
157
+ at 224x224x25, which is the smallest canvas its 7x7 kernel accepts at all. Nothing
158
+ fits - not base resolution, not the smallest clip the decoder will take. The path that does is NATTEN's `na3d` kernel, named per component as
159
+ `"attn_processor_type": "diffusers.models.autoencoders.ltx2_diffusion_decoder.LTX2VideoVaeNeighborhoodNattenProcessor"`,
160
+ which builds no mask at all - but it is fetched from the Hub by the `kernels` package
161
+ and needs a `shi-labs/natten` build matching the installed torch, which as of
162
+ torch 2.14 does not exist.
163
+ And a step that returns latents returns *audio* latents too, which nothing outside a
164
+ pipeline call can vocode - the two-stage template below feeds them back into one,
165
+ which is the only way they become sound. Nothing here ships the diffusion decoder.
166
+
167
+ Spend headroom on the two-stage flow rather than on base resolution: render at 768x448
168
+ on the eight distilled sigmas, double the video latents with the latent upsampler, then
169
+ renoise them and run the three stage-two sigmas at 1536x896 through the same pipeline.
170
+ That refine pass is what puts the detail back - the upsampler alone gives a soft 2x -
171
+ and it is the flow the model card, Lightricks' pipeline notes and the diffusers docs all
172
+ describe. The base pass keeps its pipeline loaded so the refine pass is served from the
173
+ cache; the refine pass releases it. Since 2026-08 Lightricks route production quality
174
+ through their DFR pipeline instead, which diffusers ships and nothing here uses yet.
175
+ Measured on an RTX 3090 the refined clip is sharper than the 2x upsample alone at the
176
+ same seed: fur, branches and snow texture resolve where the upsample-only frame is a
177
+ soft blur. About eight warm minutes, three and a half of them writing the full-size clip.
178
+
179
+ **Examples:** [text-to-video.json](../workflows/templates/ltx2/text-to-video.json) (t2v),
180
+ [two-stage.json](../workflows/templates/ltx2/two-stage.json) (base -> latent upsample -> refine),
181
+ [keyframes.json](../workflows/templates/ltx2/keyframes.json) (first and
182
+ last frame), [extend-clip.json](../workflows/templates/ltx2/extend-clip.json) (continue a clip),
183
+ [generative-upscale.json](../workflows/templates/ltx2/generative-upscale.json) (generative 2x upscale via IC-LoRA),
184
+ [enhance-prompt.json](../workflows/templates/ltx2/enhance-prompt.json) (native prompt
185
+ enhancer and duration head)
186
+
187
+ ## SDXL (2.6B UNet)
188
+
189
+ Fits several times over in 24GB. Skip quantization and offloading entirely; `compile` the UNet if you generate many images per session. Use `num_images_per_prompt` batching with `vae.enable_slicing`.
190
+
191
+ **Example:** [base-and-refiner.json](../workflows/templates/base-and-refiner.json)
192
+
193
+ ## Multi-model workflows
194
+
195
+ When a workflow chains two large models (generate → upscale, generate → interpolate), release the first pipeline instead of offloading everything:
196
+
197
+ ```json
198
+ { "name": "generate", "release_pipeline": true, "pipeline": { ... } }
199
+ ```
200
+
201
+ See [WORKFLOW_GUIDE.md](WORKFLOW_GUIDE.md#releasing-a-pipeline-mid-workflow).
dw/docs/RELEASING.md ADDED
@@ -0,0 +1,195 @@
1
+ # Releasing
2
+
3
+ ## Unreleased
4
+
5
+ This project has no standing release-notes file - GitHub auto-generates
6
+ notes from commits at tag time (see below). This section is a scratch pad
7
+ for items a branch's author wants the next release note to name; clear it
8
+ when a release ships.
9
+
10
+ ### 0.4.0
11
+
12
+ The auto-generated notes for this range are a single merge line, since the
13
+ work landed on `develop` without PRs. Paste this section into the GitHub
14
+ release body once the tag has published (`gh release edit v0.4.0
15
+ --notes-file ...`).
16
+
17
+ **Breaking and behaviour changes**
18
+
19
+ - `download_output` over a `dw.serve --mcp` endpoint refuses a call with no
20
+ `destination`. It used to write into the server's own directory (#353).
21
+ - Untrusted workflows are refused in more cases (#409-#413):
22
+ - a `*_type` that doesn't resolve to a class, or that isn't a kind a
23
+ workflow constructs: a diffusers or transformers model, pipeline,
24
+ scheduler, tokenizer or processor, a quantization config, an auto
25
+ factory, a diffusers reference/condition type or an attention processor.
26
+ A plain `torch` class such as `torch.nn.Linear` is now refused;
27
+ - `constant:` walks through `_` names or out of the allowed packages;
28
+ - URLs with backslashes;
29
+ - `text/html` and `text/xml` result types;
30
+ - media hosts that aren't globally routable, including 100.64/10 (CGNAT,
31
+ and so Tailscale);
32
+ - more than 5 redirects;
33
+ - images over 50M pixels.
34
+
35
+ Listings and export zips drop symlinks that escape their root.
36
+ `--trust-workflows` lifts all of these.
37
+ - `run_workflow` validates the caller's `arguments` when it queues the job
38
+ (#414/#415). `validate_workflow(arguments={})` checks a run with no values
39
+ supplied, not just the document (#364).
40
+ - A fractional value for an int variable is refused (#338), and so is a
41
+ still image passed as a video argument (#347).
42
+ - `templates/minimax/music` normalizes to -3 dBFS instead of -1, so its output
43
+ is quieter (#362).
44
+ - Every response carries `X-Content-Type-Options: nosniff` and
45
+ `X-Frame-Options: DENY`. Active document types under `/outputs` and
46
+ `/inputs` are served with `Content-Security-Policy: sandbox`.
47
+ - A validate-time probe reads only a literal media path that the run itself
48
+ would be allowed to read.
49
+ - A dict or list passed to a string-typed variable is refused (#433).
50
+ `templates/ltx2/keyframes` takes `first_image`/`last_image` as plain
51
+ strings, not `{"location": ...}` (#431/#433).
52
+ - `loop_frames` returns float32 frames in [0, 1] instead of uint8, the shape
53
+ `LTX2ReferenceCondition` needs; a keyframe condition still wants
54
+ `frames_as_array`. `ltx2/reference-sheet`'s default asset is now
55
+ `asset:reference_sheet.jpg` (#444).
56
+ - `validate_workflow` refuses a `components` name the pipeline doesn't
57
+ register; `duration_head` is gone from the in-context LTX-2 templates
58
+ (#442).
59
+ - A `{"media_type": "image"}` reference on a video argument loads as a
60
+ one-frame still (#443).
61
+ - `pair_audio fit: "video"` always fits, and warns on any nonzero gap
62
+ (#428/#429). `concat_videos` and `dissolve_videos` pad a short joined
63
+ track to the frame grid, warning (`joined_audio_padded_to_frames`) only
64
+ when the pad is a frame or more; a residual the AAC mux trims off is
65
+ logged, or warned as `joined_audio_short_after_mux` from a frame up.
66
+ `media.shots` is measured against the file as written (#426/#435/#454).
67
+ Neither warns about resampling inputs that agree to a pinned
68
+ `sample_rate` (#453).
69
+ - New warnings: `match_levels_near_silent` (#434), and `shot_span_overrun`
70
+ from the probes plus a validate-time check (#425).
71
+ - Error text changed: `delete_workspace` (#437/#438), the sub-workflow path
72
+ refusal names the places it looked (#422), and `/outputs/asset:...` misses
73
+ name the asset without server paths.
74
+
75
+ **New**
76
+
77
+ - The `assess_output` tool and `GET /api/gallery/{name}/assess`, plus the
78
+ probe tasks `analyze_shots`, `analyze_seams` and `analyze_sync_drift`
79
+ (#387/#388).
80
+ - A joined video records its shot boundaries (`media.shots`).
81
+ `get_output_frames(seams=true)` uses them, so it no longer needs
82
+ `boundaries` (#385).
83
+ - Run versions (`v<N>`):
84
+ - `list_gallery` returns `run_id`/`version` and filters by `folder` and
85
+ `version`;
86
+ - `output:<wf>/v<N>/<file>` references;
87
+ - `wait_for_job` returns `run_version`;
88
+ - export zips download as `<wf>-vN-<job>.zip`.
89
+ - `list_gallery(media=true)` adds durations, and `output:` names work in
90
+ gallery reads (#356).
91
+ - `DW_PUBLIC_URL` adds absolute URLs to gallery and export responses.
92
+ `export_job` also returns `auth_required` and `open_url` (#353).
93
+ - A `grade` task for images and video: exposure, contrast, saturation and
94
+ temperature/tint (#349).
95
+ - The `templates/minimax/shots-batch` H3 template (#352).
96
+ - Every generative template takes a `seed` argument (#351).
97
+ - `normalize_audio(target_lufs)`, and `integrated_lufs` plus true peak in
98
+ media metadata (#361).
99
+ - `gain_audio` with no region gains the whole track (#395).
100
+ - `world_fade_out_ms` on `assemble-and-score` (#339).
101
+ - Download progress shows in `phase_detail` (#343). `phase_stall` events
102
+ now read as informational (#357).
103
+ - `workflow`, `inline_workflow` and `prompt` also accept a JSON string. A
104
+ mistyped workflow name gets suggestions from the catalog (#397).
105
+ - Host caches are released when each job ends (#368), and the skills point
106
+ at `clear_memory`.
107
+ - `get_job_events(kinds=...)` and `?kinds=` on the event-log route; a kind
108
+ matches an event's `event` or its `kind`, so `["phase_stall"]` selects
109
+ one warning type (#436).
110
+ - `get_memory` reports the step cache's `entries` and `retained_bytes`
111
+ (#418).
112
+ - `get_output_image` and `/outputs` resolve `asset:` references (#445), and
113
+ `get_output_frames(seams=true)` works on linked assets (#430).
114
+ - Compact `assess_output` lists each finding once (#427). Shots are named by
115
+ their source when joined inputs already carry shots (#432).
116
+ - A task-only workflow's run history counts, so its estimate can quote
117
+ `basis: observed` (#439). The Music 3 hint no longer shows on video
118
+ (#441).
119
+
120
+ **Fixes**
121
+
122
+ - The step cache's retained-byte count no longer only grows (#418).
123
+ - `templates/ltx2/keyframes` (#431), `restore-decompression` (#442) and
124
+ `reference-sheet` (#444) run with their own defaults again.
125
+ - Joined audio and shot maps stay on the frame grid through repeated joins
126
+ (#423, #426, #428, #435).
127
+
128
+ Releases are cut by pushing a `v<semver>` tag. CI does the rest.
129
+
130
+ Before merging `develop` into `master`, run `scripts/preflight.sh` and get it
131
+ passing. It covers more than CI: ruff over the whole repo rather than
132
+ `dw dw_mcp tests`, and the UI's Playwright e2e tests, which CI doesn't run.
133
+
134
+ ```bash
135
+ scripts/release.sh 0.38.0
136
+ scripts/release.sh 0.38.0-alpha.1 "UI front end" # optional tag message
137
+ ```
138
+
139
+ The script bumps `pyproject.toml` (the single source of the version —
140
+ `dw.__version__` reads it at runtime) and sets the same version in
141
+ `plugins/dw/.claude-plugin/plugin.json`, so an installed plugin names the
142
+ engine it was written against; it commits just those two files, pushes
143
+ master, tags the bump commit `v0.38.0`, and pushes the tag. It refuses
144
+ a malformed version, a branch other than master, an existing tag, or a
145
+ dirty index (unstaged changes elsewhere are fine — the release commit
146
+ is path-limited to those two files).
147
+
148
+ By hand, the equivalent is:
149
+
150
+ ```bash
151
+ # 1. Bump the version in pyproject.toml:
152
+ # version = "0.38.0"
153
+ # 2. Set the same version in plugins/dw/.claude-plugin/plugin.json
154
+ git commit -m "release 0.38.0" -- pyproject.toml plugins/dw/.claude-plugin/plugin.json
155
+
156
+ # 3. Tag the bump commit and push
157
+ git tag -a v0.38.0 -m "release 0.38.0"
158
+ git push origin master v0.38.0
159
+ ```
160
+
161
+ The tag must point at a commit whose pyproject already declares the
162
+ same version — the release job checks and refuses a mismatch.
163
+
164
+ The tag triggers the full CI chain: backend tests, UI lint/type-check/
165
+ unit tests, then the wheel build (SPA compiled into the package via
166
+ `scripts/build_dist.sh`). Only if all of that passes does the `release`
167
+ job run — it verifies the tag matches the pyproject version, then
168
+ creates a GitHub release named after the tag with auto-generated notes
169
+ and the wheel + sdist attached.
170
+
171
+ Note on pre-release numbering: Python packaging normalizes semver-style
172
+ pre-releases, so a `0.38.0-alpha.1` version builds a wheel named
173
+ `0.38.0a1`. The tag, pyproject, and release stay in the semver form;
174
+ only the wheel filename and pip metadata show the normalized one.
175
+
176
+ A pre-release tag like `v0.38.0-rc1` is marked as a pre-release on
177
+ GitHub. Tags that aren't `v` + semver (or that don't match the declared
178
+ versions) fail the release job before anything is published.
179
+
180
+ After the GitHub release, the `pypi` job publishes the same artifacts to
181
+ PyPI via [trusted publishing](https://docs.pypi.org/trusted-publishers/)
182
+ (OIDC — no token stored anywhere). One-time setup on pypi.org under
183
+ *Publishing*: add a trusted publisher for project `diffusers-workflow`
184
+ with owner `dkackman`, repository `diffusers-workflow`, workflow
185
+ `ci.yml`, environment `pypi` (use "add a pending publisher" before the
186
+ first release, since the project won't exist yet). Pre-release versions
187
+ are hidden from plain `pip install`; they need `pip install --pre`.
188
+
189
+ Note: released `diffusers` from PyPI may lag the newest model pipelines
190
+ this project targets — a PyPI install can need
191
+ `pip install git+https://github.com/huggingface/diffusers` on top.
192
+
193
+ To rebuild artifacts without releasing, run the CI workflow manually
194
+ (`workflow_dispatch`) — the wheel job uploads `dist/*` as a workflow
195
+ artifact.