diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/content_types.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""A step's result 'content_type': the MIME type its writer is chosen by.
|
|
2
|
+
|
|
3
|
+
`content_type: "video"` validated clean and then died deep inside a writer,
|
|
4
|
+
in a traceback naming neither the field nor the value (#168) - the same
|
|
5
|
+
shape as #162 in a different field. The writer's own dispatch matches on
|
|
6
|
+
`content_type.startswith("video")`, so the bare word "video" took the video
|
|
7
|
+
branch anyway; having no real MIME type it also had no extension to write,
|
|
8
|
+
and the closest writer imageio could guess from an empty one was not a video
|
|
9
|
+
writer at all.
|
|
10
|
+
|
|
11
|
+
Audio and video each go through exactly one container this engine writes -
|
|
12
|
+
`AUDIO_FORMATS` in `dw/result.py`, and the single `video/mp4` mux - so
|
|
13
|
+
anything else in either family is refused here rather than accepted only to
|
|
14
|
+
mismatch its writer later. image/*, text/* and *.json values stay
|
|
15
|
+
permissive beyond the MIME-shape check: their writer dispatch is a generic
|
|
16
|
+
prefix/suffix match (PIL's own format inference, a literal text or JSON
|
|
17
|
+
write) with no narrower container to enforce a whitelist against.
|
|
18
|
+
|
|
19
|
+
The one text/* exception is active content. `/outputs` serves a written
|
|
20
|
+
file on the UI's own origin, without a token, so an .html or .xml output is
|
|
21
|
+
a page whose script reads the API token the UI keeps in localStorage (#407).
|
|
22
|
+
`text/html` and `text/xml` are the two active types the text writer can
|
|
23
|
+
produce, so they are refused outright; the server also serves every active
|
|
24
|
+
type it finds on disk under a `Content-Security-Policy: sandbox`, since a
|
|
25
|
+
planted file never passes through here.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from .for_each import MEMBER_SEPARATOR, render_path
|
|
29
|
+
from .result import AUDIO_FORMATS, MUXED_VIDEO_CONTENT_TYPE
|
|
30
|
+
from .security import InvalidInputError, validate_content_type
|
|
31
|
+
|
|
32
|
+
CONTENT_TYPE_KEY = "content_type"
|
|
33
|
+
|
|
34
|
+
# Reference prefixes substitution resolves before this pass runs. One still
|
|
35
|
+
# spelled out here is one nothing resolved, and that is the undeclared-
|
|
36
|
+
# variable pass's complaint rather than a shape error
|
|
37
|
+
_UNRESOLVED_PREFIXES = ("variable:", "item:")
|
|
38
|
+
|
|
39
|
+
# Result types a browser would run as a document on the UI origin
|
|
40
|
+
REFUSED_ACTIVE_CONTENT_TYPES = frozenset({"text/html", "text/xml"})
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _active_content_fault(value):
|
|
44
|
+
# compared without parameters or case: 'Text/HTML; charset=utf-8' is
|
|
45
|
+
# the same document type
|
|
46
|
+
if (
|
|
47
|
+
isinstance(value, str)
|
|
48
|
+
and value.split(";", 1)[0].strip().lower() in REFUSED_ACTIVE_CONTENT_TYPES
|
|
49
|
+
):
|
|
50
|
+
return (
|
|
51
|
+
f"Invalid content_type: {value!r} - active content is not written: "
|
|
52
|
+
f"a browser would run it as a page on the server's origin. Use "
|
|
53
|
+
f"'text/plain' or 'application/json'"
|
|
54
|
+
)
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def content_type_fault(value):
|
|
59
|
+
"""Why this result 'content_type' is invalid, or None.
|
|
60
|
+
|
|
61
|
+
Raises nothing - callers that already have an InvalidInputError-raising
|
|
62
|
+
check (validate_content_type) can call that directly; this is the
|
|
63
|
+
string-message form `content_type_errors` collects.
|
|
64
|
+
"""
|
|
65
|
+
try:
|
|
66
|
+
validate_content_type(value)
|
|
67
|
+
except InvalidInputError as e:
|
|
68
|
+
return str(e)
|
|
69
|
+
|
|
70
|
+
active = _active_content_fault(value)
|
|
71
|
+
if active is not None:
|
|
72
|
+
return active
|
|
73
|
+
main_type = value.split("/", 1)[0]
|
|
74
|
+
if main_type == "audio" and value not in AUDIO_FORMATS:
|
|
75
|
+
return (
|
|
76
|
+
f"Invalid content_type: {value!r} - audio can be written as "
|
|
77
|
+
f"{', '.join(sorted(AUDIO_FORMATS))}"
|
|
78
|
+
)
|
|
79
|
+
if main_type == "video" and value != MUXED_VIDEO_CONTENT_TYPE:
|
|
80
|
+
return (
|
|
81
|
+
f"Invalid content_type: {value!r} - video is only written as "
|
|
82
|
+
f"'{MUXED_VIDEO_CONTENT_TYPE}'"
|
|
83
|
+
)
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def content_type_errors(workflow_definition, source_indices=None):
|
|
88
|
+
"""Every result 'content_type' that no writer will accept, as
|
|
89
|
+
[{path, message}].
|
|
90
|
+
|
|
91
|
+
The definition handed here has already been substituted and expanded,
|
|
92
|
+
so every value in it is literal; a 'variable:' or 'item:' still spelled
|
|
93
|
+
out is left alone. `source_indices`, when given, is the source step
|
|
94
|
+
index of each step - a 'for_each' group turns one written step into
|
|
95
|
+
several, and the path an error carries has to be one the author can
|
|
96
|
+
find in the file they wrote; the member is named in the message.
|
|
97
|
+
"""
|
|
98
|
+
steps = workflow_definition.get("steps")
|
|
99
|
+
if not isinstance(steps, list):
|
|
100
|
+
return []
|
|
101
|
+
|
|
102
|
+
errors = []
|
|
103
|
+
for index, step in enumerate(steps):
|
|
104
|
+
if not isinstance(step, dict):
|
|
105
|
+
continue
|
|
106
|
+
result = step.get("result")
|
|
107
|
+
if not isinstance(result, dict) or CONTENT_TYPE_KEY not in result:
|
|
108
|
+
continue
|
|
109
|
+
value = result[CONTENT_TYPE_KEY]
|
|
110
|
+
if isinstance(value, str) and value.startswith(_UNRESOLVED_PREFIXES):
|
|
111
|
+
continue
|
|
112
|
+
source = (
|
|
113
|
+
source_indices[index]
|
|
114
|
+
if source_indices is not None and index < len(source_indices)
|
|
115
|
+
else index
|
|
116
|
+
)
|
|
117
|
+
name = step.get("name")
|
|
118
|
+
where = (
|
|
119
|
+
f" in member '{name}'"
|
|
120
|
+
if isinstance(name, str) and MEMBER_SEPARATOR in name
|
|
121
|
+
else ""
|
|
122
|
+
)
|
|
123
|
+
fault = content_type_fault(value)
|
|
124
|
+
if fault is not None:
|
|
125
|
+
errors.append(
|
|
126
|
+
{
|
|
127
|
+
"path": render_path(("steps", source, "result", CONTENT_TYPE_KEY)),
|
|
128
|
+
"message": f"{fault}{where}",
|
|
129
|
+
}
|
|
130
|
+
)
|
|
131
|
+
return errors
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def refuse_active_content_type(value):
|
|
135
|
+
"""Raise InvalidInputError for an active result type - the writer's
|
|
136
|
+
run-time half of the refusal `content_type_errors` makes, for a
|
|
137
|
+
definition that reached it without validation. Only this refusal: the
|
|
138
|
+
writer's own dispatch still answers every other value as it did."""
|
|
139
|
+
fault = _active_content_fault(value)
|
|
140
|
+
if fault is not None:
|
|
141
|
+
raise InvalidInputError(fault)
|
|
142
|
+
return value
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
__all__ = [
|
|
146
|
+
"REFUSED_ACTIVE_CONTENT_TYPES",
|
|
147
|
+
"content_type_errors",
|
|
148
|
+
"content_type_fault",
|
|
149
|
+
"refuse_active_content_type",
|
|
150
|
+
]
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""A `dissolve_videos` overlap too wide for one of its own inputs, refused
|
|
2
|
+
before the run when the frame counts are already knowable.
|
|
3
|
+
|
|
4
|
+
`dissolve_videos` (`dw/tasks/dissolve_videos.py`) raises once it has decoded
|
|
5
|
+
every input: a video with fewer frames than its share of `dissolve_frames`
|
|
6
|
+
overlaps (`seams * dissolve_frames`) fails with "video N has M frames, too
|
|
7
|
+
few for its S dissolve(s) of F frames". That is correct, but late - a chain
|
|
8
|
+
that generates each shot before joining them can spend many GPU minutes
|
|
9
|
+
reaching a step that was always going to fail, for an arithmetic mistake
|
|
10
|
+
visible from the workflow document alone (#400).
|
|
11
|
+
|
|
12
|
+
Moved here, into `validation_errors`, for exactly the cases where a video's
|
|
13
|
+
frame count is knowable without running anything: a literal file path inside
|
|
14
|
+
the directories the run may read (`dw/probe_paths.py`), or an
|
|
15
|
+
`asset:`/`output:` reference, with a literal `dissolve_frames`.
|
|
16
|
+
`resolve_path_references` is what turns either into a real path before the
|
|
17
|
+
run reads it; `probe_media` decodes that file the same way `dw/server/app.py`
|
|
18
|
+
already does for gallery metadata. A `previous_result:` (or any reference
|
|
19
|
+
`expand_for_each` left unresolved), a `variable:`/`item:`/`gather:` reference,
|
|
20
|
+
or a non-literal `dissolve_frames`, names no frame count yet and is left to
|
|
21
|
+
the existing run-time check - silence there is correct, not a gap, since the
|
|
22
|
+
length is not known until the step that produces it runs.
|
|
23
|
+
|
|
24
|
+
`concat_videos`'s `trim_frames` and `crossfade_audio`'s crossfade window were
|
|
25
|
+
each considered for the same treatment - the issue that motivated this module
|
|
26
|
+
asked whether they "probably have the same gap". They do not: neither raises
|
|
27
|
+
when an input is too short. `concat_videos` silently truncates
|
|
28
|
+
(`frames.extend(clip[head_trim:])`), and `crossfade_audio` silently clamps
|
|
29
|
+
its window to the shortest side (`crossfade_concat`) - a different, and
|
|
30
|
+
already silent, shape of problem with no run-time error to move earlier.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from .for_each import MEMBER_SEPARATOR, render_path
|
|
34
|
+
from .media_info import probe_media
|
|
35
|
+
from .probe_paths import resolve_probe_path
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _frame_count(path):
|
|
39
|
+
"""The frame count `dissolve_videos` would see for this file, or None
|
|
40
|
+
when it cannot be probed or carries no video stream."""
|
|
41
|
+
info = probe_media(path)
|
|
42
|
+
if info is None or info.get("kind") != "video":
|
|
43
|
+
return None
|
|
44
|
+
return info.get("frame_count")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def dissolve_frame_errors(workflow_definition, source_indices=None, base_dir=None):
|
|
48
|
+
"""Every `dissolve_videos` step whose overlap already exceeds a
|
|
49
|
+
statically-resolvable input's real frame count, as [{path, message}].
|
|
50
|
+
|
|
51
|
+
Walks the substituted, expanded definition, the same convention
|
|
52
|
+
`video_extension_errors` and `task_argument_errors` follow:
|
|
53
|
+
`source_indices` maps an expanded step back to the one the author wrote,
|
|
54
|
+
and a path inside a `for_each` member names the member.
|
|
55
|
+
"""
|
|
56
|
+
steps = workflow_definition.get("steps")
|
|
57
|
+
if not isinstance(steps, list):
|
|
58
|
+
return []
|
|
59
|
+
|
|
60
|
+
errors = []
|
|
61
|
+
for index, step in enumerate(steps):
|
|
62
|
+
if not isinstance(step, dict):
|
|
63
|
+
continue
|
|
64
|
+
task = step.get("task")
|
|
65
|
+
if not isinstance(task, dict) or task.get("command") != "dissolve_videos":
|
|
66
|
+
continue
|
|
67
|
+
arguments = task.get("arguments")
|
|
68
|
+
if not isinstance(arguments, dict):
|
|
69
|
+
continue
|
|
70
|
+
videos = arguments.get("videos")
|
|
71
|
+
if not isinstance(videos, list) or len(videos) < 2:
|
|
72
|
+
continue
|
|
73
|
+
dissolve_frames = arguments.get("dissolve_frames", 12)
|
|
74
|
+
if not isinstance(dissolve_frames, (int, float)) or isinstance(
|
|
75
|
+
dissolve_frames, bool
|
|
76
|
+
):
|
|
77
|
+
continue
|
|
78
|
+
if dissolve_frames <= 0:
|
|
79
|
+
continue
|
|
80
|
+
|
|
81
|
+
problems = []
|
|
82
|
+
for video_index, video in enumerate(videos):
|
|
83
|
+
path = resolve_probe_path(video, base_dir, "a video argument")
|
|
84
|
+
if path is None:
|
|
85
|
+
continue
|
|
86
|
+
frame_count = _frame_count(path)
|
|
87
|
+
if frame_count is None:
|
|
88
|
+
continue
|
|
89
|
+
seams = (video_index > 0) + (video_index < len(videos) - 1)
|
|
90
|
+
needed = seams * dissolve_frames
|
|
91
|
+
if frame_count < needed:
|
|
92
|
+
problems.append(
|
|
93
|
+
f"video {video_index} has {frame_count} frames, too few "
|
|
94
|
+
f"for its {seams} dissolve(s) of {dissolve_frames} frames"
|
|
95
|
+
)
|
|
96
|
+
if not problems:
|
|
97
|
+
continue
|
|
98
|
+
|
|
99
|
+
source = (
|
|
100
|
+
source_indices[index]
|
|
101
|
+
if source_indices is not None and index < len(source_indices)
|
|
102
|
+
else index
|
|
103
|
+
)
|
|
104
|
+
name = step.get("name")
|
|
105
|
+
where = (
|
|
106
|
+
f" in member '{name}'"
|
|
107
|
+
if isinstance(name, str) and MEMBER_SEPARATOR in name
|
|
108
|
+
else ""
|
|
109
|
+
)
|
|
110
|
+
errors.append(
|
|
111
|
+
{
|
|
112
|
+
"path": render_path(
|
|
113
|
+
("steps", source, "task", "arguments", "dissolve_frames")
|
|
114
|
+
),
|
|
115
|
+
"message": f"dissolve_videos: {'; '.join(problems)}{where}",
|
|
116
|
+
}
|
|
117
|
+
)
|
|
118
|
+
return errors
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
__all__ = ["dissolve_frame_errors"]
|
dw/docs/ACCELERATION.md
ADDED
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
# Inference Acceleration
|
|
2
|
+
|
|
3
|
+
Speed up generation by caching intermediate computations and skipping redundant transformer steps. Two systems are available: diffusers built-in caching and TeaCache. Beyond caching, `torch.compile`, attention backend selection, layerwise casting, and device-level settings (TF32, cuDNN) also affect throughput - see below. Memory offloading trades speed for VRAM and is covered in depth in [WORKFLOW_GUIDE.md](WORKFLOW_GUIDE.md#memory-offloading).
|
|
4
|
+
|
|
5
|
+
For ready-made configurations that combine these levers per model family, see [RECIPES_24GB.md](RECIPES_24GB.md).
|
|
6
|
+
|
|
7
|
+
## Diffusers Built-in Cache
|
|
8
|
+
|
|
9
|
+
Applied at pipeline load time via the `cache` configuration. Hooks auto-reset between runs.
|
|
10
|
+
|
|
11
|
+
### Models diffusers has not registered
|
|
12
|
+
|
|
13
|
+
`first_block`, `mag` and `layer_skip` look a model's transformer block class up in
|
|
14
|
+
diffusers' own registry and raise when it is absent, which is how a model that supports
|
|
15
|
+
`enable_cache()` ends up with no usable cache. [cache_blocks.json](../dw/cache_blocks.json)
|
|
16
|
+
fills those gaps in, registering the missing block metadata on demand; entries become
|
|
17
|
+
redundant, not wrong, once diffusers registers the same class upstream. MiniMax-H3 and
|
|
18
|
+
LTX-2 are listed there today.
|
|
19
|
+
|
|
20
|
+
LTX-2's block needs one thing more than the metadata diffusers defines. It returns two
|
|
21
|
+
streams - video and audio - and diffusers reads the second one back out of a forward
|
|
22
|
+
argument named literally `encoder_hidden_states`, the only two-stream shape it registers
|
|
23
|
+
upstream (text beside image). LTX-2's blocks take an `encoder_hidden_states` of their
|
|
24
|
+
own, the text conditioning, so the fixed name reads the wrong tensor and feeds the text
|
|
25
|
+
embeddings back as the audio stream on every skipped block. The entry names the argument
|
|
26
|
+
its second stream actually comes from (`encoder_hidden_states_argument_name`), which is
|
|
27
|
+
what makes caching correct there rather than merely quiet.
|
|
28
|
+
|
|
29
|
+
### FirstBlockCache
|
|
30
|
+
|
|
31
|
+
Simplest and broadest support. Compares first-block residuals to decide whether to skip remaining blocks.
|
|
32
|
+
|
|
33
|
+
```json
|
|
34
|
+
"configuration": {
|
|
35
|
+
"component_type": "FluxPipeline",
|
|
36
|
+
"cache": {
|
|
37
|
+
"type": "first_block",
|
|
38
|
+
"threshold": 0.05
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Higher threshold = more speedup, more quality loss. Start with `0.05` and increase to taste.
|
|
44
|
+
|
|
45
|
+
**Example:** [step-caching.json](../workflows/templates/step-caching.json)
|
|
46
|
+
|
|
47
|
+
### MagCache
|
|
48
|
+
|
|
49
|
+
Magnitude-based caching with error accumulation. Requires `num_inference_steps` to match the pipeline arguments, and `mag_ratios` — the per-step magnitude ratios, which are checkpoint-dependent:
|
|
50
|
+
|
|
51
|
+
```json
|
|
52
|
+
"cache": {
|
|
53
|
+
"type": "mag",
|
|
54
|
+
"mag_ratios": "flux",
|
|
55
|
+
"threshold": 0.06,
|
|
56
|
+
"num_inference_steps": 28,
|
|
57
|
+
"max_skip_steps": 3,
|
|
58
|
+
"retention_ratio": 0.2
|
|
59
|
+
}
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
| Property | Default | Description |
|
|
63
|
+
| -------- | ------- | ----------- |
|
|
64
|
+
| `mag_ratios` | required | Preset name or explicit per-step ratio array — see below |
|
|
65
|
+
| `threshold` | 0.06 | Accumulated error threshold for skipping |
|
|
66
|
+
| `num_inference_steps` | required | Must match pipeline arguments |
|
|
67
|
+
| `max_skip_steps` | 3 | Max consecutive steps to skip |
|
|
68
|
+
| `retention_ratio` | 0.2 | Fraction of initial steps where skipping is disabled |
|
|
69
|
+
| `calibrate` | false | Measure ratios for a new model instead of skipping — see below |
|
|
70
|
+
|
|
71
|
+
#### Supplying `mag_ratios`
|
|
72
|
+
|
|
73
|
+
MagCache needs to know how each denoising step's output magnitude typically behaves for *your* checkpoint, so unlike the other cache types it cannot run on defaults alone. Give it either:
|
|
74
|
+
|
|
75
|
+
- **A preset name** — `"mag_ratios": "flux"` resolves to the ratios diffusers ships for Flux. Any preset a later diffusers release adds is usable by name without a change here.
|
|
76
|
+
- **An explicit array** — `"mag_ratios": [1.0, 0.98, 0.96, ...]`. The array is interpolated automatically when its length differs from `num_inference_steps`, so ratios measured at one step count can be reused at another.
|
|
77
|
+
|
|
78
|
+
For a model with no preset, run once with `"calibrate": true`. Calibration skips nothing and logs the measured ratios at the end of the run; paste that array into `mag_ratios` and drop the `calibrate` flag for subsequent runs.
|
|
79
|
+
|
|
80
|
+
```json
|
|
81
|
+
"cache": { "type": "mag", "calibrate": true, "num_inference_steps": 28 }
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
### TaylorSeerCache
|
|
85
|
+
|
|
86
|
+
Taylor series approximation of cached outputs:
|
|
87
|
+
|
|
88
|
+
```json
|
|
89
|
+
"cache": {
|
|
90
|
+
"type": "taylorseer",
|
|
91
|
+
"cache_interval": 5,
|
|
92
|
+
"max_order": 1
|
|
93
|
+
}
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
| Property | Default | Description |
|
|
97
|
+
| -------- | ------- | ----------- |
|
|
98
|
+
| `cache_interval` | 5 | Full computation every N steps |
|
|
99
|
+
| `max_order` | 1 | Taylor series order (higher = better approximation, more memory) |
|
|
100
|
+
|
|
101
|
+
### FasterCache
|
|
102
|
+
|
|
103
|
+
Experimental, video-oriented. Uses FFT frequency decomposition:
|
|
104
|
+
|
|
105
|
+
```json
|
|
106
|
+
"cache": {
|
|
107
|
+
"type": "faster"
|
|
108
|
+
}
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Best for video models like CogVideoX. No additional parameters needed for basic use.
|
|
112
|
+
|
|
113
|
+
### TextKVCache
|
|
114
|
+
|
|
115
|
+
Caches the transformer's key/value projections of the (unchanging) text
|
|
116
|
+
embeddings across denoising steps, recomputing only what the latents need:
|
|
117
|
+
|
|
118
|
+
```json
|
|
119
|
+
"cache": {
|
|
120
|
+
"type": "text_kv"
|
|
121
|
+
}
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
No parameters.
|
|
125
|
+
|
|
126
|
+
## TeaCache
|
|
127
|
+
|
|
128
|
+
Training-free acceleration that monkey-patches the transformer's forward function. Uses polynomial-rescaled L1 distance to determine when to skip computation.
|
|
129
|
+
|
|
130
|
+
```json
|
|
131
|
+
"configuration": {
|
|
132
|
+
"component_type": "FluxPipeline",
|
|
133
|
+
"teacache": {
|
|
134
|
+
"rel_l1_thresh": 0.6
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
TeaCache requires `num_inference_steps` in the pipeline arguments — it needs to know the total step count.
|
|
140
|
+
|
|
141
|
+
### Configuration
|
|
142
|
+
|
|
143
|
+
| Property | Description |
|
|
144
|
+
| -------- | ----------- |
|
|
145
|
+
| `rel_l1_thresh` | Cache threshold. Model-specific defaults apply if omitted. |
|
|
146
|
+
| `coefficients` | Array of 5 polynomial coefficients. Override model defaults. |
|
|
147
|
+
| `variant` | Explicit model variant for multi-variant architectures. |
|
|
148
|
+
|
|
149
|
+
### Supported Models
|
|
150
|
+
|
|
151
|
+
Model coefficients and defaults are stored in [teacache_models.json](../dw/teacache_models.json). Currently implemented with a custom forward function:
|
|
152
|
+
|
|
153
|
+
- **Flux** (FluxTransformer2DModel) — thresholds: 0.25 (~1.5x), 0.4 (~1.8x), 0.6 (~2.0x), 0.8 (~2.25x)
|
|
154
|
+
|
|
155
|
+
Registry includes coefficients for Mochi, LTX-Video, CogVideoX, HunyuanVideo, Wan2.1, and Lumina2 (forward functions pending). For any model other than Flux, use the [diffusers built-in caches](#diffusers-built-in-cache) instead - `first_block` or `mag` cover the models the registry lists.
|
|
156
|
+
|
|
157
|
+
### Variants
|
|
158
|
+
|
|
159
|
+
Some models have multiple variants with different coefficients:
|
|
160
|
+
|
|
161
|
+
```json
|
|
162
|
+
"teacache": {
|
|
163
|
+
"rel_l1_thresh": 0.2,
|
|
164
|
+
"variant": "cogvideox_2b"
|
|
165
|
+
}
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
**Example:** [step-caching.json](../workflows/templates/step-caching.json)
|
|
169
|
+
|
|
170
|
+
## Cache vs TeaCache
|
|
171
|
+
|
|
172
|
+
| | Diffusers Cache | TeaCache |
|
|
173
|
+
| --- | --- | --- |
|
|
174
|
+
| Setup | Built into diffusers | Custom forward functions |
|
|
175
|
+
| Model support | Any transformer with CacheMixin | Requires per-model implementation |
|
|
176
|
+
| Maintenance | Maintained by HuggingFace | Maintained in this project |
|
|
177
|
+
| Configuration | Set once at load time | Applied per-execution via context manager |
|
|
178
|
+
| Approach | Various algorithms (block, magnitude, Taylor) | Polynomial-rescaled L1 distance |
|
|
179
|
+
|
|
180
|
+
They are **mutually exclusive** — use one or the other, not both.
|
|
181
|
+
|
|
182
|
+
For most cases, start with `first_block` cache. Use TeaCache when you need fine-tuned control over Flux acceleration thresholds.
|
|
183
|
+
|
|
184
|
+
## Attention Backends
|
|
185
|
+
|
|
186
|
+
Select the attention implementation diffusers uses for the duration of each pipeline call, via a context manager wrapped around `pipeline(...)`:
|
|
187
|
+
|
|
188
|
+
```json
|
|
189
|
+
"configuration": {
|
|
190
|
+
"component_type": "FluxPipeline",
|
|
191
|
+
"attention_backend": "flash_hub"
|
|
192
|
+
}
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Common values: `"flash"`, `"flash_hub"`, `"sage"`, `"sage_hub"`, `"native"`, `"flex"`. The full set is diffusers' `AttentionBackendName` enum - availability depends on what's installed (`flash-attn`, `sageattention`, etc.) and the platform. `_hub`-suffixed backends are fetched from the Hugging Face Hub kernel registry on first use, which needs the `kernels` package installed (`pip install kernels`) - it is not a dw dependency, and no bundled workflow sets a backend, so each runs on a plain install.
|
|
196
|
+
|
|
197
|
+
A component can also pin its backend persistently instead, via `set_attention_backend`:
|
|
198
|
+
|
|
199
|
+
```json
|
|
200
|
+
"configuration": {
|
|
201
|
+
"components": {
|
|
202
|
+
"transformer": { "attention_backend": "flash_hub" }
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
Prefer the pinned form for a compiled component - the per-call context manager switches implementations under the compiled graph and forces a recompile on every run.
|
|
208
|
+
|
|
209
|
+
## Attention Slicing
|
|
210
|
+
|
|
211
|
+
```json
|
|
212
|
+
"configuration": {
|
|
213
|
+
"component_type": "FluxPipeline",
|
|
214
|
+
"enable_attention_slicing": true
|
|
215
|
+
}
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
Processes attention in slices to reduce memory at some cost to speed. Enabled automatically on MPS (unified memory benefits from slicing) unless `disable_attention_slicing` is set. Modular pipelines have no `enable_attention_slicing()` method - the setting is silently skipped rather than failing when the pipeline doesn't support it.
|
|
219
|
+
|
|
220
|
+
## torch.compile
|
|
221
|
+
|
|
222
|
+
Compile a component once it is fully configured - the graph captures final dtypes, adapters, quantization, and offload hooks. Configured per component under `components`:
|
|
223
|
+
|
|
224
|
+
```json
|
|
225
|
+
"configuration": {
|
|
226
|
+
"component_type": "FluxPipeline",
|
|
227
|
+
"components": {
|
|
228
|
+
"transformer": {
|
|
229
|
+
"compile": {
|
|
230
|
+
"repeated_blocks": true,
|
|
231
|
+
"fullgraph": true
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
```
|
|
237
|
+
|
|
238
|
+
| Property | Description |
|
|
239
|
+
| -------- | ----------- |
|
|
240
|
+
| `repeated_blocks` | Compile only the model's repeated block classes (diffusers regional compilation). Near the same speedup as full compilation with a fraction of the cold-start cost. Recommended. |
|
|
241
|
+
| `mode` | torch.compile mode: `"default"`, `"reduce-overhead"`, `"max-autotune"`. |
|
|
242
|
+
| `fullgraph` | Require a single graph with no breaks - fails fast instead of silently losing speedup. |
|
|
243
|
+
| `dynamic` | Compile with dynamic shapes. Set `true` when resolutions or frame counts vary between runs to avoid recompiles. |
|
|
244
|
+
|
|
245
|
+
Typical gains are 1.3-1.5x on diffusion transformers, and compilation stacks with the caches above. Notes:
|
|
246
|
+
|
|
247
|
+
- **First run pays the compile cost.** The [REPL](REPL_COMMANDS.md)'s persistent worker keeps compiled pipelines loaded between runs, so the cost is paid once per session rather than once per generation.
|
|
248
|
+
- **Pin the attention backend** on a compiled component (`"attention_backend"` in the same `components` entry) rather than using the pipeline-level per-call context manager, which forces recompiles.
|
|
249
|
+
- **Composes with offloading**: apply `group_offload` and `compile` on the same component and the offload hooks are installed first, as required. Skipped with a warning on MPS.
|
|
250
|
+
- **Don't combine `fullgraph` with a `cache`**: the cache hooks decide skip-or-compute per step, a data-dependent branch diffusers wraps in `torch.compiler.disable` - it needs the graph break that `fullgraph: true` forbids. Compile with the default (partial) graph mode when a cache is active.
|
|
251
|
+
- **TorchAO quantization needs compile to be fast** - see [QUANTIZATION.md](QUANTIZATION.md#torchao).
|
|
252
|
+
|
|
253
|
+
**Example:** [flux-dev-compile.json](../workflows/models/flux-dev-compile.json), [flux-torchao.json](../workflows/models/flux-torchao.json)
|
|
254
|
+
|
|
255
|
+
## Layerwise Casting
|
|
256
|
+
|
|
257
|
+
Store a component's weights in a narrow dtype and upcast only for compute, per component:
|
|
258
|
+
|
|
259
|
+
```json
|
|
260
|
+
"transformer": {
|
|
261
|
+
"configuration": { "component_type": "FluxTransformer2DModel" },
|
|
262
|
+
"enable_layerwise_casting": {
|
|
263
|
+
"storage_dtype": "torch.float8_e4m3fn",
|
|
264
|
+
"compute_dtype": "torch.bfloat16"
|
|
265
|
+
},
|
|
266
|
+
"from_pretrained_arguments": { ... }
|
|
267
|
+
}
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
Both `storage_dtype` and `compute_dtype` are required. Applied via the component's own `enable_layerwise_casting()` right after it loads, so it composes with quantization and group offloading on the same component.
|
|
271
|
+
|
|
272
|
+
## Memory Format and Attention Processors
|
|
273
|
+
|
|
274
|
+
Three older per-component knobs, set in the pipeline `configuration` beside
|
|
275
|
+
`vae` / `unet` / `transformer` and applied right after the components load:
|
|
276
|
+
|
|
277
|
+
| Key | Where | Effect |
|
|
278
|
+
| --- | ----- | ------ |
|
|
279
|
+
| `channels_last` | `vae`, `unet` | `to(memory_format=torch.channels_last)`. Faster convolutions on CUDA for a convolutional UNet or VAE; nothing to gain on a transformer |
|
|
280
|
+
| `enable_forward_chunking` | `unet` | Runs the UNet's feed-forward layers in chunks - less peak memory, slightly slower |
|
|
281
|
+
| `attn_processor_type` | `unet`, `transformer` | Names an attention processor class to install with `set_attn_processor` (the name is resolved and constructed, so it goes through the `_type` conversion: `"AttnProcessor2_0"`). For a per-call backend instead, see [Attention Backends](#attention-backends) |
|
|
282
|
+
|
|
283
|
+
```json
|
|
284
|
+
"configuration": {
|
|
285
|
+
"component_type": "StableDiffusionPipeline",
|
|
286
|
+
"unet": { "channels_last": true, "enable_forward_chunking": true },
|
|
287
|
+
"vae": { "enable_slicing": true, "channels_last": true }
|
|
288
|
+
}
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
## Memory Offloading
|
|
292
|
+
|
|
293
|
+
`offload` (`"model"` or `"sequential"`) and `group_offload` trade speed for VRAM by streaming weights between system memory and the accelerator instead of keeping everything resident. `"model"` moves whole submodules and costs the least speed; `"sequential"` moves individual layers and is the slowest but uses the least memory; block/leaf-level `group_offload` sits between the two and is what a modular pipeline's self-loaded components use, since they aren't reachable in time for `offload`. Full configuration syntax is in [WORKFLOW_GUIDE.md](WORKFLOW_GUIDE.md#memory-offloading). Omit both for the fastest run, when VRAM allows it.
|
|
294
|
+
|
|
295
|
+
`"residency": "on_demand"` on a component is the cheap case of the same trade: the model rests in system memory and is moved to the device whole around each of its own calls. That is a bad deal for anything called once per step, and a good one for a VAE called twice a run - it frees the VAE's VRAM for the denoise loop at the cost of two transfers, where group offloading the same VAE would restream it once per decode tile. See [On-demand components](WORKFLOW_GUIDE.md#on-demand-components).
|
|
296
|
+
|
|
297
|
+
**Example:** [flux-dev.json](../workflows/models/flux-dev.json) (`"offload": "model"`), [z-image.json](../workflows/models/z-image.json) (`"offload": "sequential"`), [video-with-audio.json](../workflows/templates/minimax/video-with-audio.json) (`group_offload` per component), [reference-to-video.json](../workflows/templates/minimax/reference-to-video.json) (`group_offload` for the transformer, `on_demand` for the VAEs)
|
|
298
|
+
|
|
299
|
+
## Reading Memory While Offloading
|
|
300
|
+
|
|
301
|
+
A workflow that offloads keeps its weights in host memory by design, so the
|
|
302
|
+
card can sit near-empty through a generation and the VRAM figures alone say
|
|
303
|
+
nothing about what a run holds or fails to release. `get_memory` (MCP) and
|
|
304
|
+
`GET /api/memory` report both: `gpu_*` is the card, `host_memory_rss_mb` is
|
|
305
|
+
what the worker process holds and `host_memory_peak_rss_mb` the most it has
|
|
306
|
+
ever held, beside the machine's `host_memory_total_mb` /
|
|
307
|
+
`host_memory_available_mb`.
|
|
308
|
+
|
|
309
|
+
`host_pinned_reserved_mb` / `host_pinned_allocated_mb`, where the platform
|
|
310
|
+
reports them, are torch's pinned-host cache - the staging buffers group
|
|
311
|
+
offloading moves weights through. They are part of `host_memory_rss_mb` and
|
|
312
|
+
invisible in every `gpu_*` figure, so a worker that has released every model
|
|
313
|
+
and still holds gigabytes is usually holding these; they are returned when
|
|
314
|
+
the worker switches to a different workflow (#98). A host field is absent,
|
|
315
|
+
rather than null, on a platform that cannot measure it.
|
|
316
|
+
|
|
317
|
+
## TF32 and cuDNN
|
|
318
|
+
|
|
319
|
+
Device-level settings, read once at startup from `~/.diffusers_helper/settings.json`:
|
|
320
|
+
|
|
321
|
+
| Setting | Default | Effect |
|
|
322
|
+
| ------- | ------- | ------ |
|
|
323
|
+
| `enable_tf32` | `true` | Sets `torch.set_float32_matmul_precision("high")`, and on CUDA also `torch.backends.cuda.matmul.allow_tf32 = True`. ~2x faster matmuls on Ampere+ GPUs (RTX 30/40 series, A100, H100) with minor precision loss. No effect outside CUDA. |
|
|
324
|
+
| `cudnn_benchmark` | `true` | CUDA only. Autotunes cuDNN algorithm selection - fastest for a workflow with fixed input sizes, can add overhead when sizes vary run to run. |
|
|
325
|
+
| `cudnn_deterministic` | `false` | CUDA only. Set `true` to trade speed for reproducible output given the same seed. |
|
|
326
|
+
|
|
327
|
+
```json
|
|
328
|
+
{ "enable_tf32": true, "cudnn_benchmark": true, "cudnn_deterministic": false }
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
## Environment Defaults
|
|
332
|
+
|
|
333
|
+
Set automatically at import unless already present in the environment (export your own value to override):
|
|
334
|
+
|
|
335
|
+
| Variable | Default | Effect |
|
|
336
|
+
| -------- | ------- | ------ |
|
|
337
|
+
| `PYTORCH_CUDA_ALLOC_CONF` | `expandable_segments:True` | Lets the CUDA allocator grow segments instead of fragmenting fixed-size ones. Multi-step workflows churn differently-shaped allocations (generate, upscale, interpolate); fragmentation is what OOMs a card that nominally has room. |
|
|
338
|
+
| `HF_ENABLE_PARALLEL_LOADING` | `true` | Loads sharded checkpoints in parallel - faster cold starts. |
|
|
339
|
+
| `PYTORCH_MPS_HIGH_WATERMARK_RATIO` | `0.0` | MPS only - use all available unified memory. |
|
|
340
|
+
|
|
341
|
+
For faster model downloads, optionally `pip install hf_transfer` and set `HF_HUB_ENABLE_HF_TRANSFER=1`. Not enabled automatically - it bypasses the Python HTTP stack and breaks some proxy setups.
|
|
342
|
+
|
|
343
|
+
## MPS Notes
|
|
344
|
+
|
|
345
|
+
Apple Silicon has narrower acceleration support than CUDA:
|
|
346
|
+
|
|
347
|
+
- No flash-attn, no Triton, no bitsandbytes - `attention_backend` is effectively CUDA-only; use `"native"`-family backends or leave it unset on MPS. `compile` is skipped with a warning (inductor support on MPS is immature).
|
|
348
|
+
- No `torch.autocast` support - autocast-related warnings from other libraries are suppressed automatically rather than surfaced.
|
|
349
|
+
- `enable_attention_slicing` is on by default (set `disable_attention_slicing` to turn it off).
|
|
350
|
+
- `float16` produces NaN values on Apple Silicon - use `float32` or `bfloat16` for `torch_dtype` instead; dw only warns, it doesn't override the dtype for you.
|
|
351
|
+
- `PYTORCH_MPS_HIGH_WATERMARK_RATIO` defaults to `0.0` (use all unified memory) unless already set in the environment.
|
|
352
|
+
- Offloading has less benefit than on CUDA, since unified memory is already shared between CPU and GPU.
|