diffusers-workflow 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
  2. diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
  3. diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
  4. diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
  5. diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
  6. diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
  7. dw/__init__.py +440 -0
  8. dw/adapter_compatibility.py +226 -0
  9. dw/arguments.py +1231 -0
  10. dw/assessment_rules.py +159 -0
  11. dw/assets.py +130 -0
  12. dw/cache_blocks.json +16 -0
  13. dw/cache_blocks.py +146 -0
  14. dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
  15. dw/content_types.py +150 -0
  16. dw/dissolve_frame_errors.py +121 -0
  17. dw/docs/ACCELERATION.md +352 -0
  18. dw/docs/AGENT_LOOP.md +95 -0
  19. dw/docs/DEPENDENCIES.md +91 -0
  20. dw/docs/IP_ADAPTER.md +109 -0
  21. dw/docs/LORAS.md +131 -0
  22. dw/docs/MCP.md +517 -0
  23. dw/docs/PROMPT_WEIGHTING.md +78 -0
  24. dw/docs/QUANTIZATION.md +230 -0
  25. dw/docs/RECIPES_24GB.md +201 -0
  26. dw/docs/RELEASING.md +195 -0
  27. dw/docs/REMOTE.md +140 -0
  28. dw/docs/REPL_COMMANDS.md +121 -0
  29. dw/docs/REPL_WORKER_GUIDE.md +51 -0
  30. dw/docs/SECURITY.md +272 -0
  31. dw/docs/SECURITY_QUICKREF.md +112 -0
  32. dw/docs/SERVER.md +679 -0
  33. dw/docs/TASKS.md +1741 -0
  34. dw/docs/TESTING.md +71 -0
  35. dw/docs/WORKFLOW_GUIDE.md +2038 -0
  36. dw/docs/WORKSPACES.md +316 -0
  37. dw/download_watch.py +335 -0
  38. dw/elision.py +306 -0
  39. dw/events.py +275 -0
  40. dw/for_each.py +409 -0
  41. dw/host_memory.py +258 -0
  42. dw/host_memory_projection.py +230 -0
  43. dw/hub_cache.py +432 -0
  44. dw/introspection.py +1228 -0
  45. dw/kernel_availability.py +208 -0
  46. dw/locations.py +599 -0
  47. dw/log_setup.py +45 -0
  48. dw/loudness.py +82 -0
  49. dw/media_audio.py +217 -0
  50. dw/media_frames.py +367 -0
  51. dw/media_info.py +297 -0
  52. dw/pipeline_processors/chain.py +821 -0
  53. dw/pipeline_processors/config_objects.py +237 -0
  54. dw/pipeline_processors/pipeline.py +2297 -0
  55. dw/pipeline_processors/remote.py +46 -0
  56. dw/plan.py +920 -0
  57. dw/previous_results.py +411 -0
  58. dw/probe_paths.py +59 -0
  59. dw/prompt_schema.json +48 -0
  60. dw/prompt_weighting.py +378 -0
  61. dw/prompts.py +159 -0
  62. dw/realize.py +250 -0
  63. dw/reference_limits.py +215 -0
  64. dw/reference_names.py +125 -0
  65. dw/repl.py +338 -0
  66. dw/repl_commands.py +836 -0
  67. dw/repl_worker.py +159 -0
  68. dw/result.py +1720 -0
  69. dw/result_fps.py +82 -0
  70. dw/run.py +162 -0
  71. dw/runs.py +768 -0
  72. dw/scalar_result_validation.py +97 -0
  73. dw/schema.py +283 -0
  74. dw/security.py +1038 -0
  75. dw/select_validation.py +115 -0
  76. dw/serve.py +277 -0
  77. dw/server/__init__.py +2 -0
  78. dw/server/app.py +4586 -0
  79. dw/server/assess.py +132 -0
  80. dw/server/catalog_shape.py +487 -0
  81. dw/server/enhancers.py +129 -0
  82. dw/server/exports.py +480 -0
  83. dw/server/guides.py +257 -0
  84. dw/server/jobs.py +1561 -0
  85. dw/server/mcp_mount.py +95 -0
  86. dw/server/netinfo.py +124 -0
  87. dw/server/observed_cost.py +379 -0
  88. dw/server/sysinfo.py +71 -0
  89. dw/server/ui/assets/abap-08VXUWAP.js +1 -0
  90. dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
  91. dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
  92. dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
  93. dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
  94. dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
  95. dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
  96. dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
  97. dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
  98. dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
  99. dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
  100. dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
  101. dw/server/ui/assets/css-DIMkf-bt.js +3 -0
  102. dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
  103. dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
  104. dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
  105. dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
  106. dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
  107. dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
  108. dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
  109. dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
  110. dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
  111. dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
  112. dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
  113. dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
  114. dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
  115. dw/server/ui/assets/go-C-y9NEjX.js +1 -0
  116. dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
  117. dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
  118. dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
  119. dw/server/ui/assets/html-PhsdjHSr.js +1 -0
  120. dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
  121. dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
  122. dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
  123. dw/server/ui/assets/index-DgrYhQd9.js +43 -0
  124. dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
  125. dw/server/ui/assets/java-BEtHBSE6.js +1 -0
  126. dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
  127. dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
  128. dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
  129. dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
  130. dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
  131. dw/server/ui/assets/less-B9JPFI3C.js +2 -0
  132. dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
  133. dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
  134. dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
  135. dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
  136. dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
  137. dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
  138. dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
  139. dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
  140. dw/server/ui/assets/monaco--ixms01u.css +1 -0
  141. dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
  142. dw/server/ui/assets/msdax-DauUninz.js +1 -0
  143. dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
  144. dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
  145. dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
  146. dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
  147. dw/server/ui/assets/perl-oz_6vUea.js +1 -0
  148. dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
  149. dw/server/ui/assets/php-nr791fC2.js +1 -0
  150. dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
  151. dw/server/ui/assets/postiats-43DmfD33.js +1 -0
  152. dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
  153. dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
  154. dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
  155. dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
  156. dw/server/ui/assets/python-Bcn70HdC.js +1 -0
  157. dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
  158. dw/server/ui/assets/r-BwWrilGY.js +1 -0
  159. dw/server/ui/assets/razor-D1HmNnby.js +1 -0
  160. dw/server/ui/assets/redis-ClamHrr6.js +1 -0
  161. dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
  162. dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
  163. dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
  164. dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
  165. dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
  166. dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
  167. dw/server/ui/assets/scheme-BeGwcela.js +1 -0
  168. dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
  169. dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
  170. dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
  171. dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
  172. dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
  173. dw/server/ui/assets/sql-NEE52Syq.js +1 -0
  174. dw/server/ui/assets/st-DbInun42.js +1 -0
  175. dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
  176. dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
  177. dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
  178. dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
  179. dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
  180. dw/server/ui/assets/twig-De2hgUGE.js +1 -0
  181. dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
  182. dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
  183. dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
  184. dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
  185. dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
  186. dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
  187. dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
  188. dw/server/ui/index.html +17 -0
  189. dw/server/updater.py +192 -0
  190. dw/settings.py +98 -0
  191. dw/shot_span_preflight.py +116 -0
  192. dw/shots.py +359 -0
  193. dw/slice_preflight.py +148 -0
  194. dw/step.py +187 -0
  195. dw/step_cache.py +442 -0
  196. dw/subfolders.py +107 -0
  197. dw/task_domains.py +307 -0
  198. dw/tasks/assess.py +826 -0
  199. dw/tasks/audio_transcription.py +88 -0
  200. dw/tasks/audio_utils.py +1862 -0
  201. dw/tasks/background_remover.py +43 -0
  202. dw/tasks/borders.py +113 -0
  203. dw/tasks/compose_text.py +74 -0
  204. dw/tasks/concat_videos.py +300 -0
  205. dw/tasks/depth_estimator.py +54 -0
  206. dw/tasks/diffusion_upscale.py +109 -0
  207. dw/tasks/dissolve_videos.py +342 -0
  208. dw/tasks/format_messages.py +24 -0
  209. dw/tasks/gather.py +173 -0
  210. dw/tasks/grade.py +97 -0
  211. dw/tasks/image_to_text.py +43 -0
  212. dw/tasks/image_utils.py +764 -0
  213. dw/tasks/interpolate_frames.py +252 -0
  214. dw/tasks/judge.py +68 -0
  215. dw/tasks/model_cache.py +55 -0
  216. dw/tasks/pair_audio.py +268 -0
  217. dw/tasks/qr_code.py +19 -0
  218. dw/tasks/restore_faces.py +175 -0
  219. dw/tasks/rife_model.py +192 -0
  220. dw/tasks/segment.py +121 -0
  221. dw/tasks/select.py +111 -0
  222. dw/tasks/speech_generation.py +228 -0
  223. dw/tasks/stabilize.py +129 -0
  224. dw/tasks/task.py +920 -0
  225. dw/tasks/tensor_image.py +57 -0
  226. dw/tasks/text_generation.py +169 -0
  227. dw/tasks/text_sections.py +80 -0
  228. dw/tasks/upscale.py +203 -0
  229. dw/tasks/video_utils.py +624 -0
  230. dw/tasks/zoe_depth.py +71 -0
  231. dw/teacache.py +381 -0
  232. dw/teacache_models.json +99 -0
  233. dw/test.py +29 -0
  234. dw/type_helpers.py +231 -0
  235. dw/validate.py +68 -0
  236. dw/variable_constraints.py +444 -0
  237. dw/variables.py +443 -0
  238. dw/video_extensions.py +141 -0
  239. dw/vram_estimate.py +116 -0
  240. dw/worker.py +764 -0
  241. dw/workflow.py +2007 -0
  242. dw/workflow_schema.json +1346 -0
  243. dw/workflow_sources.py +383 -0
  244. dw/workflows/h3_context_ir.json +57 -0
  245. dw/workflows/test.json +31 -0
  246. dw/workspace.py +730 -0
  247. dw_mcp/__init__.py +6 -0
  248. dw_mcp/__main__.py +133 -0
  249. dw_mcp/assets.py +336 -0
  250. dw_mcp/authoring.py +114 -0
  251. dw_mcp/catalog.py +360 -0
  252. dw_mcp/client.py +486 -0
  253. dw_mcp/diagnose.py +371 -0
  254. dw_mcp/exports.py +84 -0
  255. dw_mcp/guides.py +35 -0
  256. dw_mcp/media.py +638 -0
  257. dw_mcp/models.py +97 -0
  258. dw_mcp/prompts.py +104 -0
  259. dw_mcp/server.py +1343 -0
  260. dw_mcp/workspaces.py +212 -0
dw/content_types.py ADDED
@@ -0,0 +1,150 @@
1
+ """A step's result 'content_type': the MIME type its writer is chosen by.
2
+
3
+ `content_type: "video"` validated clean and then died deep inside a writer,
4
+ in a traceback naming neither the field nor the value (#168) - the same
5
+ shape as #162 in a different field. The writer's own dispatch matches on
6
+ `content_type.startswith("video")`, so the bare word "video" took the video
7
+ branch anyway; having no real MIME type it also had no extension to write,
8
+ and the closest writer imageio could guess from an empty one was not a video
9
+ writer at all.
10
+
11
+ Audio and video each go through exactly one container this engine writes -
12
+ `AUDIO_FORMATS` in `dw/result.py`, and the single `video/mp4` mux - so
13
+ anything else in either family is refused here rather than accepted only to
14
+ mismatch its writer later. image/*, text/* and *.json values stay
15
+ permissive beyond the MIME-shape check: their writer dispatch is a generic
16
+ prefix/suffix match (PIL's own format inference, a literal text or JSON
17
+ write) with no narrower container to enforce a whitelist against.
18
+
19
+ The one text/* exception is active content. `/outputs` serves a written
20
+ file on the UI's own origin, without a token, so an .html or .xml output is
21
+ a page whose script reads the API token the UI keeps in localStorage (#407).
22
+ `text/html` and `text/xml` are the two active types the text writer can
23
+ produce, so they are refused outright; the server also serves every active
24
+ type it finds on disk under a `Content-Security-Policy: sandbox`, since a
25
+ planted file never passes through here.
26
+ """
27
+
28
+ from .for_each import MEMBER_SEPARATOR, render_path
29
+ from .result import AUDIO_FORMATS, MUXED_VIDEO_CONTENT_TYPE
30
+ from .security import InvalidInputError, validate_content_type
31
+
32
+ CONTENT_TYPE_KEY = "content_type"
33
+
34
+ # Reference prefixes substitution resolves before this pass runs. One still
35
+ # spelled out here is one nothing resolved, and that is the undeclared-
36
+ # variable pass's complaint rather than a shape error
37
+ _UNRESOLVED_PREFIXES = ("variable:", "item:")
38
+
39
+ # Result types a browser would run as a document on the UI origin
40
+ REFUSED_ACTIVE_CONTENT_TYPES = frozenset({"text/html", "text/xml"})
41
+
42
+
43
+ def _active_content_fault(value):
44
+ # compared without parameters or case: 'Text/HTML; charset=utf-8' is
45
+ # the same document type
46
+ if (
47
+ isinstance(value, str)
48
+ and value.split(";", 1)[0].strip().lower() in REFUSED_ACTIVE_CONTENT_TYPES
49
+ ):
50
+ return (
51
+ f"Invalid content_type: {value!r} - active content is not written: "
52
+ f"a browser would run it as a page on the server's origin. Use "
53
+ f"'text/plain' or 'application/json'"
54
+ )
55
+ return None
56
+
57
+
58
+ def content_type_fault(value):
59
+ """Why this result 'content_type' is invalid, or None.
60
+
61
+ Raises nothing - callers that already have an InvalidInputError-raising
62
+ check (validate_content_type) can call that directly; this is the
63
+ string-message form `content_type_errors` collects.
64
+ """
65
+ try:
66
+ validate_content_type(value)
67
+ except InvalidInputError as e:
68
+ return str(e)
69
+
70
+ active = _active_content_fault(value)
71
+ if active is not None:
72
+ return active
73
+ main_type = value.split("/", 1)[0]
74
+ if main_type == "audio" and value not in AUDIO_FORMATS:
75
+ return (
76
+ f"Invalid content_type: {value!r} - audio can be written as "
77
+ f"{', '.join(sorted(AUDIO_FORMATS))}"
78
+ )
79
+ if main_type == "video" and value != MUXED_VIDEO_CONTENT_TYPE:
80
+ return (
81
+ f"Invalid content_type: {value!r} - video is only written as "
82
+ f"'{MUXED_VIDEO_CONTENT_TYPE}'"
83
+ )
84
+ return None
85
+
86
+
87
+ def content_type_errors(workflow_definition, source_indices=None):
88
+ """Every result 'content_type' that no writer will accept, as
89
+ [{path, message}].
90
+
91
+ The definition handed here has already been substituted and expanded,
92
+ so every value in it is literal; a 'variable:' or 'item:' still spelled
93
+ out is left alone. `source_indices`, when given, is the source step
94
+ index of each step - a 'for_each' group turns one written step into
95
+ several, and the path an error carries has to be one the author can
96
+ find in the file they wrote; the member is named in the message.
97
+ """
98
+ steps = workflow_definition.get("steps")
99
+ if not isinstance(steps, list):
100
+ return []
101
+
102
+ errors = []
103
+ for index, step in enumerate(steps):
104
+ if not isinstance(step, dict):
105
+ continue
106
+ result = step.get("result")
107
+ if not isinstance(result, dict) or CONTENT_TYPE_KEY not in result:
108
+ continue
109
+ value = result[CONTENT_TYPE_KEY]
110
+ if isinstance(value, str) and value.startswith(_UNRESOLVED_PREFIXES):
111
+ continue
112
+ source = (
113
+ source_indices[index]
114
+ if source_indices is not None and index < len(source_indices)
115
+ else index
116
+ )
117
+ name = step.get("name")
118
+ where = (
119
+ f" in member '{name}'"
120
+ if isinstance(name, str) and MEMBER_SEPARATOR in name
121
+ else ""
122
+ )
123
+ fault = content_type_fault(value)
124
+ if fault is not None:
125
+ errors.append(
126
+ {
127
+ "path": render_path(("steps", source, "result", CONTENT_TYPE_KEY)),
128
+ "message": f"{fault}{where}",
129
+ }
130
+ )
131
+ return errors
132
+
133
+
134
+ def refuse_active_content_type(value):
135
+ """Raise InvalidInputError for an active result type - the writer's
136
+ run-time half of the refusal `content_type_errors` makes, for a
137
+ definition that reached it without validation. Only this refusal: the
138
+ writer's own dispatch still answers every other value as it did."""
139
+ fault = _active_content_fault(value)
140
+ if fault is not None:
141
+ raise InvalidInputError(fault)
142
+ return value
143
+
144
+
145
+ __all__ = [
146
+ "REFUSED_ACTIVE_CONTENT_TYPES",
147
+ "content_type_errors",
148
+ "content_type_fault",
149
+ "refuse_active_content_type",
150
+ ]
@@ -0,0 +1,121 @@
1
+ """A `dissolve_videos` overlap too wide for one of its own inputs, refused
2
+ before the run when the frame counts are already knowable.
3
+
4
+ `dissolve_videos` (`dw/tasks/dissolve_videos.py`) raises once it has decoded
5
+ every input: a video with fewer frames than its share of `dissolve_frames`
6
+ overlaps (`seams * dissolve_frames`) fails with "video N has M frames, too
7
+ few for its S dissolve(s) of F frames". That is correct, but late - a chain
8
+ that generates each shot before joining them can spend many GPU minutes
9
+ reaching a step that was always going to fail, for an arithmetic mistake
10
+ visible from the workflow document alone (#400).
11
+
12
+ Moved here, into `validation_errors`, for exactly the cases where a video's
13
+ frame count is knowable without running anything: a literal file path inside
14
+ the directories the run may read (`dw/probe_paths.py`), or an
15
+ `asset:`/`output:` reference, with a literal `dissolve_frames`.
16
+ `resolve_path_references` is what turns either into a real path before the
17
+ run reads it; `probe_media` decodes that file the same way `dw/server/app.py`
18
+ already does for gallery metadata. A `previous_result:` (or any reference
19
+ `expand_for_each` left unresolved), a `variable:`/`item:`/`gather:` reference,
20
+ or a non-literal `dissolve_frames`, names no frame count yet and is left to
21
+ the existing run-time check - silence there is correct, not a gap, since the
22
+ length is not known until the step that produces it runs.
23
+
24
+ `concat_videos`'s `trim_frames` and `crossfade_audio`'s crossfade window were
25
+ each considered for the same treatment - the issue that motivated this module
26
+ asked whether they "probably have the same gap". They do not: neither raises
27
+ when an input is too short. `concat_videos` silently truncates
28
+ (`frames.extend(clip[head_trim:])`), and `crossfade_audio` silently clamps
29
+ its window to the shortest side (`crossfade_concat`) - a different, and
30
+ already silent, shape of problem with no run-time error to move earlier.
31
+ """
32
+
33
+ from .for_each import MEMBER_SEPARATOR, render_path
34
+ from .media_info import probe_media
35
+ from .probe_paths import resolve_probe_path
36
+
37
+
38
+ def _frame_count(path):
39
+ """The frame count `dissolve_videos` would see for this file, or None
40
+ when it cannot be probed or carries no video stream."""
41
+ info = probe_media(path)
42
+ if info is None or info.get("kind") != "video":
43
+ return None
44
+ return info.get("frame_count")
45
+
46
+
47
+ def dissolve_frame_errors(workflow_definition, source_indices=None, base_dir=None):
48
+ """Every `dissolve_videos` step whose overlap already exceeds a
49
+ statically-resolvable input's real frame count, as [{path, message}].
50
+
51
+ Walks the substituted, expanded definition, the same convention
52
+ `video_extension_errors` and `task_argument_errors` follow:
53
+ `source_indices` maps an expanded step back to the one the author wrote,
54
+ and a path inside a `for_each` member names the member.
55
+ """
56
+ steps = workflow_definition.get("steps")
57
+ if not isinstance(steps, list):
58
+ return []
59
+
60
+ errors = []
61
+ for index, step in enumerate(steps):
62
+ if not isinstance(step, dict):
63
+ continue
64
+ task = step.get("task")
65
+ if not isinstance(task, dict) or task.get("command") != "dissolve_videos":
66
+ continue
67
+ arguments = task.get("arguments")
68
+ if not isinstance(arguments, dict):
69
+ continue
70
+ videos = arguments.get("videos")
71
+ if not isinstance(videos, list) or len(videos) < 2:
72
+ continue
73
+ dissolve_frames = arguments.get("dissolve_frames", 12)
74
+ if not isinstance(dissolve_frames, (int, float)) or isinstance(
75
+ dissolve_frames, bool
76
+ ):
77
+ continue
78
+ if dissolve_frames <= 0:
79
+ continue
80
+
81
+ problems = []
82
+ for video_index, video in enumerate(videos):
83
+ path = resolve_probe_path(video, base_dir, "a video argument")
84
+ if path is None:
85
+ continue
86
+ frame_count = _frame_count(path)
87
+ if frame_count is None:
88
+ continue
89
+ seams = (video_index > 0) + (video_index < len(videos) - 1)
90
+ needed = seams * dissolve_frames
91
+ if frame_count < needed:
92
+ problems.append(
93
+ f"video {video_index} has {frame_count} frames, too few "
94
+ f"for its {seams} dissolve(s) of {dissolve_frames} frames"
95
+ )
96
+ if not problems:
97
+ continue
98
+
99
+ source = (
100
+ source_indices[index]
101
+ if source_indices is not None and index < len(source_indices)
102
+ else index
103
+ )
104
+ name = step.get("name")
105
+ where = (
106
+ f" in member '{name}'"
107
+ if isinstance(name, str) and MEMBER_SEPARATOR in name
108
+ else ""
109
+ )
110
+ errors.append(
111
+ {
112
+ "path": render_path(
113
+ ("steps", source, "task", "arguments", "dissolve_frames")
114
+ ),
115
+ "message": f"dissolve_videos: {'; '.join(problems)}{where}",
116
+ }
117
+ )
118
+ return errors
119
+
120
+
121
+ __all__ = ["dissolve_frame_errors"]
@@ -0,0 +1,352 @@
1
+ # Inference Acceleration
2
+
3
+ Speed up generation by caching intermediate computations and skipping redundant transformer steps. Two systems are available: diffusers built-in caching and TeaCache. Beyond caching, `torch.compile`, attention backend selection, layerwise casting, and device-level settings (TF32, cuDNN) also affect throughput - see below. Memory offloading trades speed for VRAM and is covered in depth in [WORKFLOW_GUIDE.md](WORKFLOW_GUIDE.md#memory-offloading).
4
+
5
+ For ready-made configurations that combine these levers per model family, see [RECIPES_24GB.md](RECIPES_24GB.md).
6
+
7
+ ## Diffusers Built-in Cache
8
+
9
+ Applied at pipeline load time via the `cache` configuration. Hooks auto-reset between runs.
10
+
11
+ ### Models diffusers has not registered
12
+
13
+ `first_block`, `mag` and `layer_skip` look a model's transformer block class up in
14
+ diffusers' own registry and raise when it is absent, which is how a model that supports
15
+ `enable_cache()` ends up with no usable cache. [cache_blocks.json](../dw/cache_blocks.json)
16
+ fills those gaps in, registering the missing block metadata on demand; entries become
17
+ redundant, not wrong, once diffusers registers the same class upstream. MiniMax-H3 and
18
+ LTX-2 are listed there today.
19
+
20
+ LTX-2's block needs one thing more than the metadata diffusers defines. It returns two
21
+ streams - video and audio - and diffusers reads the second one back out of a forward
22
+ argument named literally `encoder_hidden_states`, the only two-stream shape it registers
23
+ upstream (text beside image). LTX-2's blocks take an `encoder_hidden_states` of their
24
+ own, the text conditioning, so the fixed name reads the wrong tensor and feeds the text
25
+ embeddings back as the audio stream on every skipped block. The entry names the argument
26
+ its second stream actually comes from (`encoder_hidden_states_argument_name`), which is
27
+ what makes caching correct there rather than merely quiet.
28
+
29
+ ### FirstBlockCache
30
+
31
+ Simplest and broadest support. Compares first-block residuals to decide whether to skip remaining blocks.
32
+
33
+ ```json
34
+ "configuration": {
35
+ "component_type": "FluxPipeline",
36
+ "cache": {
37
+ "type": "first_block",
38
+ "threshold": 0.05
39
+ }
40
+ }
41
+ ```
42
+
43
+ Higher threshold = more speedup, more quality loss. Start with `0.05` and increase to taste.
44
+
45
+ **Example:** [step-caching.json](../workflows/templates/step-caching.json)
46
+
47
+ ### MagCache
48
+
49
+ Magnitude-based caching with error accumulation. Requires `num_inference_steps` to match the pipeline arguments, and `mag_ratios` — the per-step magnitude ratios, which are checkpoint-dependent:
50
+
51
+ ```json
52
+ "cache": {
53
+ "type": "mag",
54
+ "mag_ratios": "flux",
55
+ "threshold": 0.06,
56
+ "num_inference_steps": 28,
57
+ "max_skip_steps": 3,
58
+ "retention_ratio": 0.2
59
+ }
60
+ ```
61
+
62
+ | Property | Default | Description |
63
+ | -------- | ------- | ----------- |
64
+ | `mag_ratios` | required | Preset name or explicit per-step ratio array — see below |
65
+ | `threshold` | 0.06 | Accumulated error threshold for skipping |
66
+ | `num_inference_steps` | required | Must match pipeline arguments |
67
+ | `max_skip_steps` | 3 | Max consecutive steps to skip |
68
+ | `retention_ratio` | 0.2 | Fraction of initial steps where skipping is disabled |
69
+ | `calibrate` | false | Measure ratios for a new model instead of skipping — see below |
70
+
71
+ #### Supplying `mag_ratios`
72
+
73
+ MagCache needs to know how each denoising step's output magnitude typically behaves for *your* checkpoint, so unlike the other cache types it cannot run on defaults alone. Give it either:
74
+
75
+ - **A preset name** — `"mag_ratios": "flux"` resolves to the ratios diffusers ships for Flux. Any preset a later diffusers release adds is usable by name without a change here.
76
+ - **An explicit array** — `"mag_ratios": [1.0, 0.98, 0.96, ...]`. The array is interpolated automatically when its length differs from `num_inference_steps`, so ratios measured at one step count can be reused at another.
77
+
78
+ For a model with no preset, run once with `"calibrate": true`. Calibration skips nothing and logs the measured ratios at the end of the run; paste that array into `mag_ratios` and drop the `calibrate` flag for subsequent runs.
79
+
80
+ ```json
81
+ "cache": { "type": "mag", "calibrate": true, "num_inference_steps": 28 }
82
+ ```
83
+
84
+ ### TaylorSeerCache
85
+
86
+ Taylor series approximation of cached outputs:
87
+
88
+ ```json
89
+ "cache": {
90
+ "type": "taylorseer",
91
+ "cache_interval": 5,
92
+ "max_order": 1
93
+ }
94
+ ```
95
+
96
+ | Property | Default | Description |
97
+ | -------- | ------- | ----------- |
98
+ | `cache_interval` | 5 | Full computation every N steps |
99
+ | `max_order` | 1 | Taylor series order (higher = better approximation, more memory) |
100
+
101
+ ### FasterCache
102
+
103
+ Experimental, video-oriented. Uses FFT frequency decomposition:
104
+
105
+ ```json
106
+ "cache": {
107
+ "type": "faster"
108
+ }
109
+ ```
110
+
111
+ Best for video models like CogVideoX. No additional parameters needed for basic use.
112
+
113
+ ### TextKVCache
114
+
115
+ Caches the transformer's key/value projections of the (unchanging) text
116
+ embeddings across denoising steps, recomputing only what the latents need:
117
+
118
+ ```json
119
+ "cache": {
120
+ "type": "text_kv"
121
+ }
122
+ ```
123
+
124
+ No parameters.
125
+
126
+ ## TeaCache
127
+
128
+ Training-free acceleration that monkey-patches the transformer's forward function. Uses polynomial-rescaled L1 distance to determine when to skip computation.
129
+
130
+ ```json
131
+ "configuration": {
132
+ "component_type": "FluxPipeline",
133
+ "teacache": {
134
+ "rel_l1_thresh": 0.6
135
+ }
136
+ }
137
+ ```
138
+
139
+ TeaCache requires `num_inference_steps` in the pipeline arguments — it needs to know the total step count.
140
+
141
+ ### Configuration
142
+
143
+ | Property | Description |
144
+ | -------- | ----------- |
145
+ | `rel_l1_thresh` | Cache threshold. Model-specific defaults apply if omitted. |
146
+ | `coefficients` | Array of 5 polynomial coefficients. Override model defaults. |
147
+ | `variant` | Explicit model variant for multi-variant architectures. |
148
+
149
+ ### Supported Models
150
+
151
+ Model coefficients and defaults are stored in [teacache_models.json](../dw/teacache_models.json). Currently implemented with a custom forward function:
152
+
153
+ - **Flux** (FluxTransformer2DModel) — thresholds: 0.25 (~1.5x), 0.4 (~1.8x), 0.6 (~2.0x), 0.8 (~2.25x)
154
+
155
+ Registry includes coefficients for Mochi, LTX-Video, CogVideoX, HunyuanVideo, Wan2.1, and Lumina2 (forward functions pending). For any model other than Flux, use the [diffusers built-in caches](#diffusers-built-in-cache) instead - `first_block` or `mag` cover the models the registry lists.
156
+
157
+ ### Variants
158
+
159
+ Some models have multiple variants with different coefficients:
160
+
161
+ ```json
162
+ "teacache": {
163
+ "rel_l1_thresh": 0.2,
164
+ "variant": "cogvideox_2b"
165
+ }
166
+ ```
167
+
168
+ **Example:** [step-caching.json](../workflows/templates/step-caching.json)
169
+
170
+ ## Cache vs TeaCache
171
+
172
+ | | Diffusers Cache | TeaCache |
173
+ | --- | --- | --- |
174
+ | Setup | Built into diffusers | Custom forward functions |
175
+ | Model support | Any transformer with CacheMixin | Requires per-model implementation |
176
+ | Maintenance | Maintained by HuggingFace | Maintained in this project |
177
+ | Configuration | Set once at load time | Applied per-execution via context manager |
178
+ | Approach | Various algorithms (block, magnitude, Taylor) | Polynomial-rescaled L1 distance |
179
+
180
+ They are **mutually exclusive** — use one or the other, not both.
181
+
182
+ For most cases, start with `first_block` cache. Use TeaCache when you need fine-tuned control over Flux acceleration thresholds.
183
+
184
+ ## Attention Backends
185
+
186
+ Select the attention implementation diffusers uses for the duration of each pipeline call, via a context manager wrapped around `pipeline(...)`:
187
+
188
+ ```json
189
+ "configuration": {
190
+ "component_type": "FluxPipeline",
191
+ "attention_backend": "flash_hub"
192
+ }
193
+ ```
194
+
195
+ Common values: `"flash"`, `"flash_hub"`, `"sage"`, `"sage_hub"`, `"native"`, `"flex"`. The full set is diffusers' `AttentionBackendName` enum - availability depends on what's installed (`flash-attn`, `sageattention`, etc.) and the platform. `_hub`-suffixed backends are fetched from the Hugging Face Hub kernel registry on first use, which needs the `kernels` package installed (`pip install kernels`) - it is not a dw dependency, and no bundled workflow sets a backend, so each runs on a plain install.
196
+
197
+ A component can also pin its backend persistently instead, via `set_attention_backend`:
198
+
199
+ ```json
200
+ "configuration": {
201
+ "components": {
202
+ "transformer": { "attention_backend": "flash_hub" }
203
+ }
204
+ }
205
+ ```
206
+
207
+ Prefer the pinned form for a compiled component - the per-call context manager switches implementations under the compiled graph and forces a recompile on every run.
208
+
209
+ ## Attention Slicing
210
+
211
+ ```json
212
+ "configuration": {
213
+ "component_type": "FluxPipeline",
214
+ "enable_attention_slicing": true
215
+ }
216
+ ```
217
+
218
+ Processes attention in slices to reduce memory at some cost to speed. Enabled automatically on MPS (unified memory benefits from slicing) unless `disable_attention_slicing` is set. Modular pipelines have no `enable_attention_slicing()` method - the setting is silently skipped rather than failing when the pipeline doesn't support it.
219
+
220
+ ## torch.compile
221
+
222
+ Compile a component once it is fully configured - the graph captures final dtypes, adapters, quantization, and offload hooks. Configured per component under `components`:
223
+
224
+ ```json
225
+ "configuration": {
226
+ "component_type": "FluxPipeline",
227
+ "components": {
228
+ "transformer": {
229
+ "compile": {
230
+ "repeated_blocks": true,
231
+ "fullgraph": true
232
+ }
233
+ }
234
+ }
235
+ }
236
+ ```
237
+
238
+ | Property | Description |
239
+ | -------- | ----------- |
240
+ | `repeated_blocks` | Compile only the model's repeated block classes (diffusers regional compilation). Near the same speedup as full compilation with a fraction of the cold-start cost. Recommended. |
241
+ | `mode` | torch.compile mode: `"default"`, `"reduce-overhead"`, `"max-autotune"`. |
242
+ | `fullgraph` | Require a single graph with no breaks - fails fast instead of silently losing speedup. |
243
+ | `dynamic` | Compile with dynamic shapes. Set `true` when resolutions or frame counts vary between runs to avoid recompiles. |
244
+
245
+ Typical gains are 1.3-1.5x on diffusion transformers, and compilation stacks with the caches above. Notes:
246
+
247
+ - **First run pays the compile cost.** The [REPL](REPL_COMMANDS.md)'s persistent worker keeps compiled pipelines loaded between runs, so the cost is paid once per session rather than once per generation.
248
+ - **Pin the attention backend** on a compiled component (`"attention_backend"` in the same `components` entry) rather than using the pipeline-level per-call context manager, which forces recompiles.
249
+ - **Composes with offloading**: apply `group_offload` and `compile` on the same component and the offload hooks are installed first, as required. Skipped with a warning on MPS.
250
+ - **Don't combine `fullgraph` with a `cache`**: the cache hooks decide skip-or-compute per step, a data-dependent branch diffusers wraps in `torch.compiler.disable` - it needs the graph break that `fullgraph: true` forbids. Compile with the default (partial) graph mode when a cache is active.
251
+ - **TorchAO quantization needs compile to be fast** - see [QUANTIZATION.md](QUANTIZATION.md#torchao).
252
+
253
+ **Example:** [flux-dev-compile.json](../workflows/models/flux-dev-compile.json), [flux-torchao.json](../workflows/models/flux-torchao.json)
254
+
255
+ ## Layerwise Casting
256
+
257
+ Store a component's weights in a narrow dtype and upcast only for compute, per component:
258
+
259
+ ```json
260
+ "transformer": {
261
+ "configuration": { "component_type": "FluxTransformer2DModel" },
262
+ "enable_layerwise_casting": {
263
+ "storage_dtype": "torch.float8_e4m3fn",
264
+ "compute_dtype": "torch.bfloat16"
265
+ },
266
+ "from_pretrained_arguments": { ... }
267
+ }
268
+ ```
269
+
270
+ Both `storage_dtype` and `compute_dtype` are required. Applied via the component's own `enable_layerwise_casting()` right after it loads, so it composes with quantization and group offloading on the same component.
271
+
272
+ ## Memory Format and Attention Processors
273
+
274
+ Three older per-component knobs, set in the pipeline `configuration` beside
275
+ `vae` / `unet` / `transformer` and applied right after the components load:
276
+
277
+ | Key | Where | Effect |
278
+ | --- | ----- | ------ |
279
+ | `channels_last` | `vae`, `unet` | `to(memory_format=torch.channels_last)`. Faster convolutions on CUDA for a convolutional UNet or VAE; nothing to gain on a transformer |
280
+ | `enable_forward_chunking` | `unet` | Runs the UNet's feed-forward layers in chunks - less peak memory, slightly slower |
281
+ | `attn_processor_type` | `unet`, `transformer` | Names an attention processor class to install with `set_attn_processor` (the name is resolved and constructed, so it goes through the `_type` conversion: `"AttnProcessor2_0"`). For a per-call backend instead, see [Attention Backends](#attention-backends) |
282
+
283
+ ```json
284
+ "configuration": {
285
+ "component_type": "StableDiffusionPipeline",
286
+ "unet": { "channels_last": true, "enable_forward_chunking": true },
287
+ "vae": { "enable_slicing": true, "channels_last": true }
288
+ }
289
+ ```
290
+
291
+ ## Memory Offloading
292
+
293
+ `offload` (`"model"` or `"sequential"`) and `group_offload` trade speed for VRAM by streaming weights between system memory and the accelerator instead of keeping everything resident. `"model"` moves whole submodules and costs the least speed; `"sequential"` moves individual layers and is the slowest but uses the least memory; block/leaf-level `group_offload` sits between the two and is what a modular pipeline's self-loaded components use, since they aren't reachable in time for `offload`. Full configuration syntax is in [WORKFLOW_GUIDE.md](WORKFLOW_GUIDE.md#memory-offloading). Omit both for the fastest run, when VRAM allows it.
294
+
295
+ `"residency": "on_demand"` on a component is the cheap case of the same trade: the model rests in system memory and is moved to the device whole around each of its own calls. That is a bad deal for anything called once per step, and a good one for a VAE called twice a run - it frees the VAE's VRAM for the denoise loop at the cost of two transfers, where group offloading the same VAE would restream it once per decode tile. See [On-demand components](WORKFLOW_GUIDE.md#on-demand-components).
296
+
297
+ **Example:** [flux-dev.json](../workflows/models/flux-dev.json) (`"offload": "model"`), [z-image.json](../workflows/models/z-image.json) (`"offload": "sequential"`), [video-with-audio.json](../workflows/templates/minimax/video-with-audio.json) (`group_offload` per component), [reference-to-video.json](../workflows/templates/minimax/reference-to-video.json) (`group_offload` for the transformer, `on_demand` for the VAEs)
298
+
299
+ ## Reading Memory While Offloading
300
+
301
+ A workflow that offloads keeps its weights in host memory by design, so the
302
+ card can sit near-empty through a generation and the VRAM figures alone say
303
+ nothing about what a run holds or fails to release. `get_memory` (MCP) and
304
+ `GET /api/memory` report both: `gpu_*` is the card, `host_memory_rss_mb` is
305
+ what the worker process holds and `host_memory_peak_rss_mb` the most it has
306
+ ever held, beside the machine's `host_memory_total_mb` /
307
+ `host_memory_available_mb`.
308
+
309
+ `host_pinned_reserved_mb` / `host_pinned_allocated_mb`, where the platform
310
+ reports them, are torch's pinned-host cache - the staging buffers group
311
+ offloading moves weights through. They are part of `host_memory_rss_mb` and
312
+ invisible in every `gpu_*` figure, so a worker that has released every model
313
+ and still holds gigabytes is usually holding these; they are returned when
314
+ the worker switches to a different workflow (#98). A host field is absent,
315
+ rather than null, on a platform that cannot measure it.
316
+
317
+ ## TF32 and cuDNN
318
+
319
+ Device-level settings, read once at startup from `~/.diffusers_helper/settings.json`:
320
+
321
+ | Setting | Default | Effect |
322
+ | ------- | ------- | ------ |
323
+ | `enable_tf32` | `true` | Sets `torch.set_float32_matmul_precision("high")`, and on CUDA also `torch.backends.cuda.matmul.allow_tf32 = True`. ~2x faster matmuls on Ampere+ GPUs (RTX 30/40 series, A100, H100) with minor precision loss. No effect outside CUDA. |
324
+ | `cudnn_benchmark` | `true` | CUDA only. Autotunes cuDNN algorithm selection - fastest for a workflow with fixed input sizes, can add overhead when sizes vary run to run. |
325
+ | `cudnn_deterministic` | `false` | CUDA only. Set `true` to trade speed for reproducible output given the same seed. |
326
+
327
+ ```json
328
+ { "enable_tf32": true, "cudnn_benchmark": true, "cudnn_deterministic": false }
329
+ ```
330
+
331
+ ## Environment Defaults
332
+
333
+ Set automatically at import unless already present in the environment (export your own value to override):
334
+
335
+ | Variable | Default | Effect |
336
+ | -------- | ------- | ------ |
337
+ | `PYTORCH_CUDA_ALLOC_CONF` | `expandable_segments:True` | Lets the CUDA allocator grow segments instead of fragmenting fixed-size ones. Multi-step workflows churn differently-shaped allocations (generate, upscale, interpolate); fragmentation is what OOMs a card that nominally has room. |
338
+ | `HF_ENABLE_PARALLEL_LOADING` | `true` | Loads sharded checkpoints in parallel - faster cold starts. |
339
+ | `PYTORCH_MPS_HIGH_WATERMARK_RATIO` | `0.0` | MPS only - use all available unified memory. |
340
+
341
+ For faster model downloads, optionally `pip install hf_transfer` and set `HF_HUB_ENABLE_HF_TRANSFER=1`. Not enabled automatically - it bypasses the Python HTTP stack and breaks some proxy setups.
342
+
343
+ ## MPS Notes
344
+
345
+ Apple Silicon has narrower acceleration support than CUDA:
346
+
347
+ - No flash-attn, no Triton, no bitsandbytes - `attention_backend` is effectively CUDA-only; use `"native"`-family backends or leave it unset on MPS. `compile` is skipped with a warning (inductor support on MPS is immature).
348
+ - No `torch.autocast` support - autocast-related warnings from other libraries are suppressed automatically rather than surfaced.
349
+ - `enable_attention_slicing` is on by default (set `disable_attention_slicing` to turn it off).
350
+ - `float16` produces NaN values on Apple Silicon - use `float32` or `bfloat16` for `torch_dtype` instead; dw only warns, it doesn't override the dtype for you.
351
+ - `PYTORCH_MPS_HIGH_WATERMARK_RATIO` defaults to `0.0` (use all unified memory) unless already set in the environment.
352
+ - Offloading has less benefit than on CUDA, since unified memory is already shared between CPU and GPU.