diffusers-workflow 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
  2. diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
  3. diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
  4. diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
  5. diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
  6. diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
  7. dw/__init__.py +440 -0
  8. dw/adapter_compatibility.py +226 -0
  9. dw/arguments.py +1231 -0
  10. dw/assessment_rules.py +159 -0
  11. dw/assets.py +130 -0
  12. dw/cache_blocks.json +16 -0
  13. dw/cache_blocks.py +146 -0
  14. dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
  15. dw/content_types.py +150 -0
  16. dw/dissolve_frame_errors.py +121 -0
  17. dw/docs/ACCELERATION.md +352 -0
  18. dw/docs/AGENT_LOOP.md +95 -0
  19. dw/docs/DEPENDENCIES.md +91 -0
  20. dw/docs/IP_ADAPTER.md +109 -0
  21. dw/docs/LORAS.md +131 -0
  22. dw/docs/MCP.md +517 -0
  23. dw/docs/PROMPT_WEIGHTING.md +78 -0
  24. dw/docs/QUANTIZATION.md +230 -0
  25. dw/docs/RECIPES_24GB.md +201 -0
  26. dw/docs/RELEASING.md +195 -0
  27. dw/docs/REMOTE.md +140 -0
  28. dw/docs/REPL_COMMANDS.md +121 -0
  29. dw/docs/REPL_WORKER_GUIDE.md +51 -0
  30. dw/docs/SECURITY.md +272 -0
  31. dw/docs/SECURITY_QUICKREF.md +112 -0
  32. dw/docs/SERVER.md +679 -0
  33. dw/docs/TASKS.md +1741 -0
  34. dw/docs/TESTING.md +71 -0
  35. dw/docs/WORKFLOW_GUIDE.md +2038 -0
  36. dw/docs/WORKSPACES.md +316 -0
  37. dw/download_watch.py +335 -0
  38. dw/elision.py +306 -0
  39. dw/events.py +275 -0
  40. dw/for_each.py +409 -0
  41. dw/host_memory.py +258 -0
  42. dw/host_memory_projection.py +230 -0
  43. dw/hub_cache.py +432 -0
  44. dw/introspection.py +1228 -0
  45. dw/kernel_availability.py +208 -0
  46. dw/locations.py +599 -0
  47. dw/log_setup.py +45 -0
  48. dw/loudness.py +82 -0
  49. dw/media_audio.py +217 -0
  50. dw/media_frames.py +367 -0
  51. dw/media_info.py +297 -0
  52. dw/pipeline_processors/chain.py +821 -0
  53. dw/pipeline_processors/config_objects.py +237 -0
  54. dw/pipeline_processors/pipeline.py +2297 -0
  55. dw/pipeline_processors/remote.py +46 -0
  56. dw/plan.py +920 -0
  57. dw/previous_results.py +411 -0
  58. dw/probe_paths.py +59 -0
  59. dw/prompt_schema.json +48 -0
  60. dw/prompt_weighting.py +378 -0
  61. dw/prompts.py +159 -0
  62. dw/realize.py +250 -0
  63. dw/reference_limits.py +215 -0
  64. dw/reference_names.py +125 -0
  65. dw/repl.py +338 -0
  66. dw/repl_commands.py +836 -0
  67. dw/repl_worker.py +159 -0
  68. dw/result.py +1720 -0
  69. dw/result_fps.py +82 -0
  70. dw/run.py +162 -0
  71. dw/runs.py +768 -0
  72. dw/scalar_result_validation.py +97 -0
  73. dw/schema.py +283 -0
  74. dw/security.py +1038 -0
  75. dw/select_validation.py +115 -0
  76. dw/serve.py +277 -0
  77. dw/server/__init__.py +2 -0
  78. dw/server/app.py +4586 -0
  79. dw/server/assess.py +132 -0
  80. dw/server/catalog_shape.py +487 -0
  81. dw/server/enhancers.py +129 -0
  82. dw/server/exports.py +480 -0
  83. dw/server/guides.py +257 -0
  84. dw/server/jobs.py +1561 -0
  85. dw/server/mcp_mount.py +95 -0
  86. dw/server/netinfo.py +124 -0
  87. dw/server/observed_cost.py +379 -0
  88. dw/server/sysinfo.py +71 -0
  89. dw/server/ui/assets/abap-08VXUWAP.js +1 -0
  90. dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
  91. dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
  92. dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
  93. dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
  94. dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
  95. dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
  96. dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
  97. dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
  98. dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
  99. dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
  100. dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
  101. dw/server/ui/assets/css-DIMkf-bt.js +3 -0
  102. dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
  103. dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
  104. dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
  105. dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
  106. dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
  107. dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
  108. dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
  109. dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
  110. dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
  111. dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
  112. dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
  113. dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
  114. dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
  115. dw/server/ui/assets/go-C-y9NEjX.js +1 -0
  116. dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
  117. dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
  118. dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
  119. dw/server/ui/assets/html-PhsdjHSr.js +1 -0
  120. dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
  121. dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
  122. dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
  123. dw/server/ui/assets/index-DgrYhQd9.js +43 -0
  124. dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
  125. dw/server/ui/assets/java-BEtHBSE6.js +1 -0
  126. dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
  127. dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
  128. dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
  129. dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
  130. dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
  131. dw/server/ui/assets/less-B9JPFI3C.js +2 -0
  132. dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
  133. dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
  134. dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
  135. dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
  136. dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
  137. dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
  138. dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
  139. dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
  140. dw/server/ui/assets/monaco--ixms01u.css +1 -0
  141. dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
  142. dw/server/ui/assets/msdax-DauUninz.js +1 -0
  143. dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
  144. dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
  145. dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
  146. dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
  147. dw/server/ui/assets/perl-oz_6vUea.js +1 -0
  148. dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
  149. dw/server/ui/assets/php-nr791fC2.js +1 -0
  150. dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
  151. dw/server/ui/assets/postiats-43DmfD33.js +1 -0
  152. dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
  153. dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
  154. dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
  155. dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
  156. dw/server/ui/assets/python-Bcn70HdC.js +1 -0
  157. dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
  158. dw/server/ui/assets/r-BwWrilGY.js +1 -0
  159. dw/server/ui/assets/razor-D1HmNnby.js +1 -0
  160. dw/server/ui/assets/redis-ClamHrr6.js +1 -0
  161. dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
  162. dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
  163. dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
  164. dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
  165. dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
  166. dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
  167. dw/server/ui/assets/scheme-BeGwcela.js +1 -0
  168. dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
  169. dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
  170. dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
  171. dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
  172. dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
  173. dw/server/ui/assets/sql-NEE52Syq.js +1 -0
  174. dw/server/ui/assets/st-DbInun42.js +1 -0
  175. dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
  176. dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
  177. dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
  178. dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
  179. dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
  180. dw/server/ui/assets/twig-De2hgUGE.js +1 -0
  181. dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
  182. dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
  183. dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
  184. dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
  185. dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
  186. dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
  187. dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
  188. dw/server/ui/index.html +17 -0
  189. dw/server/updater.py +192 -0
  190. dw/settings.py +98 -0
  191. dw/shot_span_preflight.py +116 -0
  192. dw/shots.py +359 -0
  193. dw/slice_preflight.py +148 -0
  194. dw/step.py +187 -0
  195. dw/step_cache.py +442 -0
  196. dw/subfolders.py +107 -0
  197. dw/task_domains.py +307 -0
  198. dw/tasks/assess.py +826 -0
  199. dw/tasks/audio_transcription.py +88 -0
  200. dw/tasks/audio_utils.py +1862 -0
  201. dw/tasks/background_remover.py +43 -0
  202. dw/tasks/borders.py +113 -0
  203. dw/tasks/compose_text.py +74 -0
  204. dw/tasks/concat_videos.py +300 -0
  205. dw/tasks/depth_estimator.py +54 -0
  206. dw/tasks/diffusion_upscale.py +109 -0
  207. dw/tasks/dissolve_videos.py +342 -0
  208. dw/tasks/format_messages.py +24 -0
  209. dw/tasks/gather.py +173 -0
  210. dw/tasks/grade.py +97 -0
  211. dw/tasks/image_to_text.py +43 -0
  212. dw/tasks/image_utils.py +764 -0
  213. dw/tasks/interpolate_frames.py +252 -0
  214. dw/tasks/judge.py +68 -0
  215. dw/tasks/model_cache.py +55 -0
  216. dw/tasks/pair_audio.py +268 -0
  217. dw/tasks/qr_code.py +19 -0
  218. dw/tasks/restore_faces.py +175 -0
  219. dw/tasks/rife_model.py +192 -0
  220. dw/tasks/segment.py +121 -0
  221. dw/tasks/select.py +111 -0
  222. dw/tasks/speech_generation.py +228 -0
  223. dw/tasks/stabilize.py +129 -0
  224. dw/tasks/task.py +920 -0
  225. dw/tasks/tensor_image.py +57 -0
  226. dw/tasks/text_generation.py +169 -0
  227. dw/tasks/text_sections.py +80 -0
  228. dw/tasks/upscale.py +203 -0
  229. dw/tasks/video_utils.py +624 -0
  230. dw/tasks/zoe_depth.py +71 -0
  231. dw/teacache.py +381 -0
  232. dw/teacache_models.json +99 -0
  233. dw/test.py +29 -0
  234. dw/type_helpers.py +231 -0
  235. dw/validate.py +68 -0
  236. dw/variable_constraints.py +444 -0
  237. dw/variables.py +443 -0
  238. dw/video_extensions.py +141 -0
  239. dw/vram_estimate.py +116 -0
  240. dw/worker.py +764 -0
  241. dw/workflow.py +2007 -0
  242. dw/workflow_schema.json +1346 -0
  243. dw/workflow_sources.py +383 -0
  244. dw/workflows/h3_context_ir.json +57 -0
  245. dw/workflows/test.json +31 -0
  246. dw/workspace.py +730 -0
  247. dw_mcp/__init__.py +6 -0
  248. dw_mcp/__main__.py +133 -0
  249. dw_mcp/assets.py +336 -0
  250. dw_mcp/authoring.py +114 -0
  251. dw_mcp/catalog.py +360 -0
  252. dw_mcp/client.py +486 -0
  253. dw_mcp/diagnose.py +371 -0
  254. dw_mcp/exports.py +84 -0
  255. dw_mcp/guides.py +35 -0
  256. dw_mcp/media.py +638 -0
  257. dw_mcp/models.py +97 -0
  258. dw_mcp/prompts.py +104 -0
  259. dw_mcp/server.py +1343 -0
  260. dw_mcp/workspaces.py +212 -0
@@ -0,0 +1,1346 @@
1
+ {
2
+ "$id": "https://github.com/dkackman/diffusers-helper/workflow",
3
+ "$schema": "https://json-schema.org/draft/2020-12/schema#",
4
+ "description": "The definition of the diffusers-workflow.",
5
+ "type": "object",
6
+ "properties": {
7
+ "id": {
8
+ "type": "string"
9
+ },
10
+ "description": {
11
+ "type": "string"
12
+ },
13
+ "configures": {
14
+ "type": "string",
15
+ "description": "For a workflow under models/: the templates/ workflow this is a tuned per-checkpoint configuration of, as a catalog name such as 'templates/text-to-image'."
16
+ },
17
+ "shape": {
18
+ "type": "string",
19
+ "enum": ["image", "image-set", "image-edit", "shot", "sequence", "audio", "text", "utility"],
20
+ "description": "What the workflow produces. Derived from the steps by the server; declare it only where the derivation is wrong. Closed vocabulary - adding a value is additive, renaming one breaks every client that filtered on it."
21
+ },
22
+ "traits": {
23
+ "type": "array",
24
+ "uniqueItems": true,
25
+ "items": {
26
+ "type": "string",
27
+ "enum": ["has-audio", "chained", "image-conditioned", "identity-referenced", "needs-input-media", "composes-workflows"]
28
+ },
29
+ "description": "How the output is made or what it needs supplied. Derived by the server; declare only to override."
30
+ },
31
+ "summary": {
32
+ "type": "string",
33
+ "maxLength": 120,
34
+ "description": "One line saying what the workflow is for, carried in catalog listings. Defaults to the first sentence of 'description'."
35
+ },
36
+ "cost": {
37
+ "type": "array",
38
+ "description": "Measured runs, one per device the maintainer measured on. Never derived; absent means unknown. 'minutes' is the whole run's wall clock on a worker that has to load the models - what a consumer actually waits through, not the denoise loop alone; on a video template the two differ by minutes.",
39
+ "items": {
40
+ "type": "object",
41
+ "required": ["device", "vram_gb", "minutes"],
42
+ "properties": {
43
+ "device": {"type": "string", "enum": ["cuda", "mps", "cpu"]},
44
+ "name": {"type": "string", "description": "The accelerator, for a person: 'RTX 4090', 'M2 Ultra'"},
45
+ "vram_gb": {"type": "number", "minimum": 0},
46
+ "minutes": {"type": "number", "minimum": 0},
47
+ "per_entry": {
48
+ "type": "object",
49
+ "description": "For a list-driven workflow: the measured cost of one entry of 'variable' (every member it produces), and the length of the default list 'minutes' was measured with, so a run over N entries is (minutes - per_entry.minutes * entries) + per_entry.minutes * N.",
50
+ "required": ["variable", "minutes", "entries"],
51
+ "properties": {
52
+ "variable": {"type": "string"},
53
+ "minutes": {"type": "number", "minimum": 0},
54
+ "entries": {"type": "integer", "minimum": 1}
55
+ },
56
+ "additionalProperties": false
57
+ }
58
+ },
59
+ "additionalProperties": false
60
+ }
61
+ },
62
+ "cost_drivers": {
63
+ "description": "The variables that move this workflow's cost, by name - a frame count, a step count, a segment count, not a prompt or a seed. Runs of this workflow are bucketed by these values when the server reports what its own history observed, so a 141-frame run never informs a 124-frame figure. Each name must be a variable this workflow declares. Declaring none is not neutral: the observed figure then falls back to runs that overrode nothing at all, which most real runs do not.",
64
+ "type": "array",
65
+ "items": {"type": "string"},
66
+ "uniqueItems": true
67
+ },
68
+ "variables": {
69
+ "$ref": "#/$defs/arguments"
70
+ },
71
+ "vram_estimate": {
72
+ "description": "A declared VRAM ceiling over this workflow's cost_drivers, checked against each 'cost' entry's 'vram_gb' in validate_workflow. Never derived from a profiler - a maintainer's own reading of what a stage needs, the same way 'cost' itself is. required_gb = base_gb + bytes_per_voxel * (product of voxel_variables' values) / 2^30; a combination projecting above any cost entry's vram_gb for that entry's device is refused, not warned, since the failure this guards is an OOM partway through a run that already spent minutes loading. 'voxel_variables' names the cost_drivers whose product drives the estimate - typically width, height and num_frames for a video template.",
73
+ "type": "object",
74
+ "required": ["base_gb", "bytes_per_voxel", "voxel_variables"],
75
+ "properties": {
76
+ "base_gb": {"type": "number", "minimum": 0, "description": "What stays resident regardless of the voxel count - everything the pipeline keeps on the device throughout, measured or bounded by a real run."},
77
+ "bytes_per_voxel": {"type": "number", "minimum": 0, "description": "Bytes of additional VRAM per unit of the voxel_variables' product, calibrated from a real run's reported requirement at a known combination."},
78
+ "voxel_variables": {
79
+ "type": "array",
80
+ "items": {"type": "string"},
81
+ "minItems": 1,
82
+ "uniqueItems": true,
83
+ "description": "Variable names whose product is the voxel count the formula scales with. Each must be a declared variable."
84
+ },
85
+ "reason": {"type": "string"}
86
+ },
87
+ "additionalProperties": false
88
+ },
89
+ "variable_constraints": {
90
+ "description": "What each variable's value is allowed to be, by variable name. A rule the engine could only enforce after the weights were loaded costs minutes to discover; declared here it is a free refusal at validation time and a line in the catalog beside the default.",
91
+ "type": "object",
92
+ "additionalProperties": {
93
+ "$ref": "#/$defs/variable_constraint"
94
+ }
95
+ },
96
+ "seed": {
97
+ "description": "Default seed for the entire workflow. Accepts a 'variable:' reference.",
98
+ "type": [
99
+ "integer",
100
+ "string"
101
+ ],
102
+ "pattern": "^variable:",
103
+ "format": "int64"
104
+ },
105
+ "argument_template": {
106
+ "description": "Engine-injected: the arguments a parent workflow passed to this one when it ran it as a sub-workflow. Written by create_step_action from the step's 'arguments' block, not authored - a workflow file carrying one is read, but a sub-workflow step is how they are meant to be supplied.",
107
+ "type": "object"
108
+ },
109
+ "steps": {
110
+ "type": "array",
111
+ "minItems": 1,
112
+ "items": {
113
+ "$ref": "#/$defs/step"
114
+ }
115
+ }
116
+ },
117
+ "additionalProperties": false,
118
+ "required": [
119
+ "id",
120
+ "steps"
121
+ ],
122
+ "$defs": {
123
+ "variable_constraint": {
124
+ "description": "A rule the author declares for one variable's value, in the same field names a chain step's 'frame_snap' uses, so a model's frame rule is written once. Checked before anything is queued and reported beside the variable's default, so a consumer reads the rule rather than guessing it (#96).",
125
+ "type": "object",
126
+ "properties": {
127
+ "modulus": {
128
+ "type": "integer",
129
+ "minimum": 1
130
+ },
131
+ "remainder": {
132
+ "type": "integer",
133
+ "minimum": 0
134
+ },
135
+ "min_frames": {
136
+ "type": "integer",
137
+ "minimum": 1
138
+ },
139
+ "max_frames": {
140
+ "type": "integer",
141
+ "minimum": 1
142
+ },
143
+ "snap": {
144
+ "description": "What to do with a value the rule refuses. 'up' rounds to the next value the rule accepts, and warns that it did; absent, the value is refused.",
145
+ "enum": [
146
+ "up"
147
+ ]
148
+ },
149
+ "reason": {
150
+ "description": "Why the rule exists, in the author's words - quoted in the refusal and reported beside the default.",
151
+ "type": "string"
152
+ }
153
+ },
154
+ "dependentRequired": {
155
+ "modulus": [
156
+ "remainder"
157
+ ],
158
+ "remainder": [
159
+ "modulus"
160
+ ]
161
+ },
162
+ "additionalProperties": false
163
+ },
164
+ "image": {
165
+ "type": "object",
166
+ "properties": {
167
+ "location": {
168
+ "type": "string",
169
+ "format": "uri"
170
+ },
171
+ "size": {
172
+ "type": "object",
173
+ "properties": {
174
+ "width": {
175
+ "type": "integer",
176
+ "format": "int16"
177
+ },
178
+ "height": {
179
+ "type": "integer",
180
+ "format": "int16"
181
+ }
182
+ }
183
+ }
184
+ },
185
+ "required": [
186
+ "location"
187
+ ]
188
+ },
189
+ "step": {
190
+ "type": "object",
191
+ "additionalProperties": false,
192
+ "$comment": "A step is closed: the engine reads only the properties above, so an invented control-flow key ('when', 'retry') or a mistyped real one ('relase_pipeline') is a hard error rather than a silent no-op.",
193
+ "properties": {
194
+ "name": {
195
+ "type": "string"
196
+ },
197
+ "for_each": {
198
+ "description": "Run this step once per entry of a list. A list, or a 'variable:' reference to one. Inside the step, 'item:' is the entry and 'item:field' one of its fields; a later step reads every member's result with 'gather:<step name>'. Members are named '<step name>@<entry name>' (or '@<index>' for an entry without a name), so '@' is reserved in step names.",
199
+ "type": ["array", "string"],
200
+ "pattern": "^variable:",
201
+ "minItems": 1,
202
+ "maxItems": 32
203
+ },
204
+ "seed": {
205
+ "description": "Default seed for the entire step. Accepts a 'variable:' reference.",
206
+ "type": [
207
+ "integer",
208
+ "string"
209
+ ],
210
+ "pattern": "^variable:",
211
+ "format": "int64"
212
+ },
213
+ "release_pipeline": {
214
+ "description": "Unload this step's pipeline once the step completes, freeing its memory for later steps. A later pipeline_reference to this step is an error, and the REPL's cross-run cache will not retain it.",
215
+ "type": "boolean"
216
+ },
217
+ "release_models": {
218
+ "description": "Unload every cached task model once the step completes, freeing their memory for later steps. Task models otherwise stay loaded for the life of the process; set this on a task or sub-workflow step whose model is not needed again, such as a prompt expander running ahead of a generation step. A later step using the same model reloads it.",
219
+ "type": "boolean"
220
+ },
221
+ "task": {
222
+ "$ref": "#/$defs/task"
223
+ },
224
+ "pipeline": {
225
+ "$ref": "#/$defs/pipeline"
226
+ },
227
+ "pipeline_reference": {
228
+ "$ref": "#/$defs/pipeline_reference"
229
+ },
230
+ "workflow": {
231
+ "$ref": "#/$defs/workflow_reference"
232
+ },
233
+ "result": {
234
+ "$ref": "#/$defs/result"
235
+ }
236
+ },
237
+ "oneOf": [
238
+ {
239
+ "required": [
240
+ "name",
241
+ "task"
242
+ ]
243
+ },
244
+ {
245
+ "required": [
246
+ "name",
247
+ "pipeline"
248
+ ]
249
+ },
250
+ {
251
+ "required": [
252
+ "name",
253
+ "pipeline_reference"
254
+ ]
255
+ },
256
+ {
257
+ "required": [
258
+ "name",
259
+ "workflow"
260
+ ]
261
+ }
262
+ ]
263
+ },
264
+ "arguments": {
265
+ "type": "object",
266
+ "additionalProperties": {
267
+ "description": "null declares an argument that is optional - a workflow can expose a variable a caller may pass without inventing a sentinel value for its absence. If a step feeds a required task argument from a variable whose value is null, the document itself still saves and validates - but a run (or a validate/save call that supplies its own arguments) that leaves the variable null fails, since the argument the step needs was never actually given a value.",
268
+ "type": [
269
+ "string",
270
+ "integer",
271
+ "number",
272
+ "object",
273
+ "array",
274
+ "boolean",
275
+ "null"
276
+ ]
277
+ }
278
+ },
279
+ "pipeline_reference": {
280
+ "type": "object",
281
+ "additionalProperties": false,
282
+ "$comment": "Closed - see 'step'.",
283
+ "properties": {
284
+ "reference_name": {
285
+ "type": "string"
286
+ },
287
+ "chain": {
288
+ "$ref": "#/$defs/chain"
289
+ },
290
+ "arguments": {
291
+ "$ref": "#/$defs/arguments"
292
+ }
293
+ },
294
+ "required": [
295
+ "reference_name",
296
+ "arguments"
297
+ ]
298
+ },
299
+ "chain": {
300
+ "description": "Run this pipeline repeatedly, carrying visual continuity from each segment into the next, and stitch the segments into one long video. Specify the length with exactly one of 'segments' or 'match_audio'.",
301
+ "type": "object",
302
+ "properties": {
303
+ "segments": {
304
+ "description": "Number of segments to generate and stitch. Accepts a 'variable:' reference.",
305
+ "type": [
306
+ "integer",
307
+ "string"
308
+ ],
309
+ "minimum": 1
310
+ },
311
+ "match_audio": {
312
+ "description": "Derive the segment count from the duration of the audio reference in this step's arguments, slicing it into frame-aligned per-segment chunks; the final video is muxed with the original, unsliced audio track.",
313
+ "type": "boolean"
314
+ },
315
+ "continuity": {
316
+ "description": "How continuity is carried from one segment into the next. 'last_frame' carries the segment's final frame as the next one's keyframe. 'last_segment' carries the segment itself - its frames and the soundtrack generated with them - as a video reference, which keeps motion, camera and voice across the seam instead of appearance alone; it needs a 'segment_argument' holding a references list.",
317
+ "type": "string",
318
+ "enum": [
319
+ "last_frame",
320
+ "last_segment"
321
+ ],
322
+ "default": "last_frame"
323
+ },
324
+ "segment_argument": {
325
+ "description": "The pipeline argument the carry-over lands in - e.g. 'image' for image-to-video pipelines, or 'references' for reference-conditioned modular pipelines, where the carry-over is appended as an image or video reference.",
326
+ "type": "string",
327
+ "default": "image"
328
+ },
329
+ "carry_frames": {
330
+ "description": "'last_segment' only - carry just this many frames from the tail of each segment rather than all of them, cutting the reference's soundtrack to the same span. A shorter carry costs less sequence length and memory (MiniMax H3 conditions on reference clips of 2 seconds and up, i.e. 48 frames at its 24 fps). Accepts a 'variable:' reference.",
331
+ "type": [
332
+ "integer",
333
+ "string"
334
+ ],
335
+ "minimum": 1
336
+ },
337
+ "carry_audio": {
338
+ "description": "'last_segment' only - whether the carried video reference brings the soundtrack generated with it, which is what carries a voice across the seam. Turn it off when the segments are already conditioned on a supplied audio track, as in a 'match_audio' chain.",
339
+ "type": "boolean",
340
+ "default": true
341
+ },
342
+ "trim_frames": {
343
+ "description": "Frames removed from the head of every segment after the first (an image-to-video segment reproduces its keyframe as frame 0). Also bounds the audio crossfade window to trim_frames / fps seconds. Accepts a 'variable:' reference.",
344
+ "type": [
345
+ "integer",
346
+ "string"
347
+ ],
348
+ "minimum": 0,
349
+ "default": 1
350
+ },
351
+ "crossfade_ms": {
352
+ "description": "Equal-power crossfade applied to generated audio at segment boundaries, clamped to the trim_frames / fps window. Accepts a 'variable:' reference.",
353
+ "type": [
354
+ "number",
355
+ "string"
356
+ ],
357
+ "minimum": 0,
358
+ "default": 75
359
+ },
360
+ "fps": {
361
+ "description": "Frame rate used for the chain's audio math - trimming, crossfading, and match_audio planning. Defaults to the pipeline's 'frame_rate' argument when present; pipelines with a fixed rate (MiniMax H3: 24) need it set here. Accepts a 'variable:' reference.",
362
+ "type": [
363
+ "number",
364
+ "string"
365
+ ],
366
+ "exclusiveMinimum": 0
367
+ },
368
+ "frame_snap": {
369
+ "description": "Constraint the pipeline places on num_frames - counts must equal modulus * n + remainder within the bounds (MiniMax H3: modulus 17, remainder 5, min 124, max 345). Used to snap the final match_audio segment to a valid length. May instead be the string 'constraint:<variable>', naming an entry of the workflow's 'variable_constraints' - so the rule is stated once rather than twice in one file.",
370
+ "oneOf": [
371
+ {
372
+ "type": "object",
373
+ "properties": {
374
+ "modulus": {
375
+ "type": "integer",
376
+ "minimum": 1
377
+ },
378
+ "remainder": {
379
+ "type": "integer",
380
+ "minimum": 0
381
+ },
382
+ "min_frames": {
383
+ "type": "integer",
384
+ "minimum": 1
385
+ },
386
+ "max_frames": {
387
+ "type": "integer",
388
+ "minimum": 1
389
+ }
390
+ },
391
+ "required": [
392
+ "modulus",
393
+ "remainder"
394
+ ],
395
+ "additionalProperties": false
396
+ },
397
+ {
398
+ "type": "string",
399
+ "pattern": "^constraint:[a-zA-Z_][a-zA-Z0-9_-]*$"
400
+ }
401
+ ]
402
+ },
403
+ "prompts": {
404
+ "description": "Per-segment prompt overrides; segment i uses prompts[min(i, len(prompts) - 1)].",
405
+ "type": "array",
406
+ "items": {
407
+ "type": "string"
408
+ },
409
+ "minItems": 1
410
+ },
411
+ "save_segments": {
412
+ "description": "Write each completed segment to the output directory as a playable mp4 and free its frames, bounding memory to one segment - a crashed chain leaves the finished segments behind. The final video is streamed from the segment files. Requires PyAV and a frame rate.",
413
+ "type": "boolean",
414
+ "default": false
415
+ },
416
+ "keep_segments": {
417
+ "description": "Leave the segment files in place after the final video is written (default: they are removed once it saves successfully). Only meaningful with save_segments.",
418
+ "type": "boolean",
419
+ "default": false
420
+ }
421
+ },
422
+ "oneOf": [
423
+ {
424
+ "required": [
425
+ "segments"
426
+ ]
427
+ },
428
+ {
429
+ "required": [
430
+ "match_audio"
431
+ ]
432
+ }
433
+ ],
434
+ "additionalProperties": false
435
+ },
436
+ "pipeline": {
437
+ "type": "object",
438
+ "properties": {
439
+ "configuration": {
440
+ "$ref": "#/$defs/pipeline_configuration"
441
+ },
442
+ "shared_components": {
443
+ "$ref": "#/$defs/shared_components"
444
+ },
445
+ "reused_components": {
446
+ "$ref": "#/$defs/reused_components"
447
+ },
448
+ "scheduler": {
449
+ "$ref": "#/$defs/scheduler"
450
+ },
451
+ "audio_scheduler": {
452
+ "$ref": "#/$defs/scheduler"
453
+ },
454
+ "model": {
455
+ "$ref": "#/$defs/pipeline_component"
456
+ },
457
+ "transformer": {
458
+ "$ref": "#/$defs/pipeline_component"
459
+ },
460
+ "transformer_2": {
461
+ "$ref": "#/$defs/pipeline_component"
462
+ },
463
+ "vae": {
464
+ "$ref": "#/$defs/pipeline_component"
465
+ },
466
+ "unet": {
467
+ "$ref": "#/$defs/pipeline_component"
468
+ },
469
+ "text_encoder": {
470
+ "$ref": "#/$defs/pipeline_component"
471
+ },
472
+ "text_encoder_2": {
473
+ "$ref": "#/$defs/pipeline_component"
474
+ },
475
+ "text_encoder_3": {
476
+ "$ref": "#/$defs/pipeline_component"
477
+ },
478
+ "tokenizer": {
479
+ "$ref": "#/$defs/pipeline_component"
480
+ },
481
+ "tokenizer_2": {
482
+ "$ref": "#/$defs/pipeline_component"
483
+ },
484
+ "tokenizer_3": {
485
+ "$ref": "#/$defs/pipeline_component"
486
+ },
487
+ "image_encoder": {
488
+ "$ref": "#/$defs/pipeline_component"
489
+ },
490
+ "feature_extractor": {
491
+ "$ref": "#/$defs/pipeline_component"
492
+ },
493
+ "prompt_enhancer_head": {
494
+ "$ref": "#/$defs/pipeline_component"
495
+ },
496
+ "controlnet": {
497
+ "$ref": "#/$defs/controlnet"
498
+ },
499
+ "loras": {
500
+ "type": "array",
501
+ "items": {
502
+ "$ref": "#/$defs/lora"
503
+ }
504
+ },
505
+ "ip_adapter": {
506
+ "$ref": "#/$defs/ip_adapter"
507
+ },
508
+ "from_pretrained_arguments": {
509
+ "$ref": "#/$defs/from_pretrained_arguments"
510
+ },
511
+ "remote_text_encoder": {
512
+ "description": "Encode prompts with a remote text-encoder service instead of loading the text encoder locally. The pipeline's text_encoder is set to None, the prompt is POSTed to the service (authenticated with the HuggingFace token), and the returned embeddings are passed to the pipeline as prompt_embeds. Mutually exclusive with prompt_weighting.",
513
+ "type": "object",
514
+ "properties": {
515
+ "url": {
516
+ "description": "URL of the remote text encoder endpoint, e.g. a HuggingFace inference endpoint.",
517
+ "type": "string"
518
+ }
519
+ },
520
+ "required": [
521
+ "url"
522
+ ]
523
+ },
524
+ "seed": {
525
+ "description": "The seed for this pipeline. Accepts a 'variable:' reference.",
526
+ "type": [
527
+ "integer",
528
+ "string"
529
+ ],
530
+ "pattern": "^variable:",
531
+ "format": "int64"
532
+ },
533
+ "chain": {
534
+ "$ref": "#/$defs/chain"
535
+ },
536
+ "arguments": {
537
+ "$ref": "#/$defs/arguments"
538
+ }
539
+ },
540
+ "additionalProperties": {
541
+ "description": "Any other key names a pipeline component to load. Diffusers grows component names faster than this schema does, so a key whose value is a component definition - an object carrying 'from_pretrained_arguments' - is loaded under that name. Anything else is a mistyped or misplaced key and is refused rather than ignored.",
542
+ "unknownPropertyMessage": "the engine reads the properties this object declares, plus any other key whose value is a component definition (an object carrying 'from_pretrained_arguments') - anything else would be silently ignored",
543
+ "type": "object",
544
+ "required": [
545
+ "from_pretrained_arguments"
546
+ ],
547
+ "allOf": [
548
+ {
549
+ "$ref": "#/$defs/pipeline_component"
550
+ }
551
+ ]
552
+ },
553
+ "required": [
554
+ "configuration",
555
+ "from_pretrained_arguments",
556
+ "arguments"
557
+ ]
558
+ },
559
+ "pipeline_configuration": {
560
+ "type": "object",
561
+ "properties": {
562
+ "offload": {
563
+ "type": "string",
564
+ "enum": [
565
+ "model",
566
+ "sequential"
567
+ ]
568
+ },
569
+ "exclude_from_cpu_offload": {
570
+ "description": "Component names (e.g. 'vae', 'text_encoder') to keep resident on the accelerator instead of offloading. Only applies when offload is 'sequential'.",
571
+ "type": "array",
572
+ "items": {
573
+ "type": "string"
574
+ }
575
+ },
576
+ "group_offload": {
577
+ "$ref": "#/$defs/group_offload"
578
+ },
579
+ "device": {
580
+ "description": "The device to run this pipeline on, e.g. 'cuda', 'cuda:1', 'mps' or 'cpu'. Defaults to the device dw is running on and becomes the default for this pipeline's components. A backend this machine does not have is translated to the one it does, with a warning, so a workflow written on a CUDA box runs on a Mac and back again - the index is dropped in that translation. A 'cpu' device is never translated.",
581
+ "type": "string"
582
+ },
583
+ "component_type": {
584
+ "description": "The python type of pipeline to use in the format 'module.typename' module defaults to diffusers",
585
+ "type": "string"
586
+ },
587
+ "no_generator": {
588
+ "description": "Whether to use a generator for the pipeline. Some pipelines do not support generators.",
589
+ "type": "boolean"
590
+ },
591
+ "enable_attention_slicing": {
592
+ "description": "Whether to enable attention slicing for the pipeline to reduce memory usage.",
593
+ "type": "boolean"
594
+ },
595
+ "disable_attention_slicing": {
596
+ "description": "Whether to disable automatic attention slicing on MPS devices. Attention slicing is enabled by default on MPS to reduce memory usage.",
597
+ "type": "boolean"
598
+ },
599
+ "attention_backend": {
600
+ "description": "The attention backend to use for the pipeline.",
601
+ "type": "string"
602
+ },
603
+ "prompt_weighting": {
604
+ "description": "Enable A1111-style prompt weighting syntax: (word:1.5) for emphasis, [word] for de-emphasis, ((word)) for nested weighting. Also supports prompts longer than 77 tokens. Currently supports Flux pipelines. Mutually exclusive with remote_text_encoder.",
605
+ "type": "boolean"
606
+ },
607
+ "pre_load_modules": {
608
+ "description": "List of Python modules to import before loading the pipeline. Used for modules that register with diffusers on import (e.g., sdnq).",
609
+ "type": "array",
610
+ "items": {
611
+ "type": "string"
612
+ }
613
+ },
614
+ "load_components": {
615
+ "description": "Arguments passed to a modular pipeline's load_components() - modular pipelines load their component weights separately from their config. Use 'dtype' to set the component dtype and 'names' to load a subset of components.",
616
+ "type": "object",
617
+ "properties": {
618
+ "names": {
619
+ "description": "Names of the components to load. All components are loaded when omitted.",
620
+ "type": "array",
621
+ "items": {
622
+ "type": "string"
623
+ }
624
+ },
625
+ "dtype": {
626
+ "description": "The torch dtype to load components in, e.g. 'torch.bfloat16'",
627
+ "type": "string"
628
+ },
629
+ "quantization_config": {
630
+ "description": "Quantization configuration per component, keyed by component name (e.g. 'transformer', 'text_encoder'). A component not named here loads unquantized.",
631
+ "type": "object",
632
+ "additionalProperties": {
633
+ "$ref": "#/$defs/quantization_config"
634
+ }
635
+ }
636
+ }
637
+ },
638
+ "configs": {
639
+ "description": "Values a modular pipeline's blocks declare and read while they run - not components, and not call arguments. Keys are whatever the pipeline itself declares, so what may be set here depends on the model: MiniMax-H3 declares 'canvas_short_edge', 'canvas_max_pixels' and 'reference_image_short_edge'. A name the pipeline does not declare is an error rather than a setting that quietly did nothing.",
640
+ "type": "object",
641
+ "additionalProperties": {
642
+ "type": [
643
+ "string",
644
+ "integer",
645
+ "number",
646
+ "boolean",
647
+ "array",
648
+ "null"
649
+ ]
650
+ }
651
+ },
652
+ "components": {
653
+ "description": "Configuration applied to components once the pipeline has loaded them. This is where a modular pipeline's components are placed, since it pulls their weights itself. Keys name a component, optionally as a dotted path into one, e.g. 'text_encoder.model'.",
654
+ "type": "object",
655
+ "additionalProperties": {
656
+ "type": "object",
657
+ "properties": {
658
+ "group_offload": {
659
+ "$ref": "#/$defs/group_offload"
660
+ },
661
+ "device": {
662
+ "description": "The device to move this component to, e.g. 'cuda'. Only for components small enough to stay resident - the offloaded ones are placed by their own hooks.",
663
+ "type": "string"
664
+ },
665
+ "residency": {
666
+ "description": "Whether the component stays on its device for the whole run ('resident', the default) or rests in system memory and is moved to the device only while one of its own calls runs ('on_demand'). Use on_demand for a component that is large but called a handful of times, like a VAE that only encodes references and decodes the result - it frees the device for the rest of the run and, unlike group_offload, moves the model whole, so a tiled decode costs one pair of transfers rather than one per tile. Not for a component called every step, and cannot be combined with group_offload.",
667
+ "type": "string",
668
+ "enum": [
669
+ "resident",
670
+ "on_demand"
671
+ ]
672
+ },
673
+ "enable_tiling": {
674
+ "description": "Enable tiled decoding on this component, for a decoder that is not the one named 'vae' (which the pipeline-level 'vae' block covers) - LTX-2.5's 'diffusion_decoder', say, which otherwise decodes the whole video volume in one allocation. 'true' uses the model's own default tile size; an object passes tile and stride sizes through to enable_tiling(), which is what a card smaller than those defaults needs.",
675
+ "type": [
676
+ "boolean",
677
+ "object"
678
+ ]
679
+ },
680
+ "attention_backend": {
681
+ "description": "Pin this component's attention backend persistently via set_attention_backend, e.g. 'flash', 'sage', '_flash_3_hub'. Use this instead of the pipeline-level attention_backend when the component is compiled - the per-call context manager forces recompiles.",
682
+ "type": "string"
683
+ },
684
+ "attn_processor_type": {
685
+ "description": "The attention processor this component runs, by type name, constructed with no arguments and handed to set_attn_processor(). The pipeline-level 'unet' and 'transformer' blocks cover those two components; this covers any other one that carries attention - LTX-2.5's 'diffusion_decoder', whose default processor is a portable FlexAttention fallback rather than the NATTEN path the decoder was built around.",
686
+ "type": "string"
687
+ },
688
+ "truncate_layers": {
689
+ "description": "Drop the tail of a ModuleList inside this component that the run never reads, before any offload hooks are installed. Keys are dotted paths relative to the component, values the number of entries to keep. For an encoder used for its hidden states: MiniMax-H3 conditions on hidden_states[50] of its 64-layer Qwen3-VL, so { 'language_model.layers': 51 } drops the 13 layers whose output nothing consumes while leaving hidden_states[50] bit-identical (keeping only 50 would hand back the final-norm output instead, a different tensor).",
690
+ "type": "object",
691
+ "additionalProperties": {
692
+ "type": "integer",
693
+ "minimum": 1
694
+ }
695
+ },
696
+ "remove_modules": {
697
+ "description": "Dotted paths to modules inside this component that the run never calls, each replaced with an Identity before any offload hooks are installed - a language-model head on a model used as an encoder, say. The attribute survives for anything that looks it up; the weights are freed rather than held (and, offloaded, pinned) for a call that never comes.",
698
+ "type": "array",
699
+ "items": {
700
+ "type": "string"
701
+ }
702
+ },
703
+ "compile": {
704
+ "$ref": "#/$defs/compile_config"
705
+ }
706
+ }
707
+ }
708
+ },
709
+ "preserve_device_placement": {
710
+ "description": "Whether to leave the component where it loaded rather than moving it to the device. Needed when the component is loaded already placed - by a 'device_map', or by a quantization that pins its tensors to one device. A 'components' block that group offloads already keeps the pipeline off the device on its own. Renamed from 'do_not_send_to_device', which is no longer recognized.",
711
+ "type": "boolean"
712
+ },
713
+ "components_manager": {
714
+ "description": "Attach a ComponentsManager to a modular pipeline. The manager tracks the pipeline's components and, with auto CPU offload enabled, keeps only the running ones on the device. It replaces 'offload', which modular pipelines do not support.",
715
+ "type": "object",
716
+ "properties": {
717
+ "enable_auto_cpu_offload": {
718
+ "description": "Move components between the device and system memory as the pipeline runs. Requires a device that reports free memory (CUDA). When enabled the manager owns device placement, so the pipeline is not moved to the device directly.",
719
+ "type": "boolean",
720
+ "default": false
721
+ },
722
+ "memory_reserve_margin": {
723
+ "description": "Device memory to keep free when deciding what to offload, e.g. '3GB'",
724
+ "type": "string",
725
+ "default": "3GB"
726
+ }
727
+ }
728
+ },
729
+ "sdnq_optimize": {
730
+ "description": "List of pipeline component names to apply SDNQ quantized matmul optimization to (e.g., ['transformer', 'text_encoder']). Requires sdnq package and CUDA/XPU hardware.",
731
+ "type": "array",
732
+ "items": {
733
+ "type": "string"
734
+ }
735
+ },
736
+ "cache": {
737
+ "description": "Enable diffusers built-in cache acceleration on the transformer. Mutually exclusive with teacache. Hooks auto-reset between inference runs.",
738
+ "type": "object",
739
+ "properties": {
740
+ "type": {
741
+ "description": "Cache algorithm: first_block (simplest, broadest support), faster (video-oriented, experimental), mag (magnitude-based, needs num_inference_steps), taylorseer (Taylor series approximation), text_kv (text key-value cache).",
742
+ "type": "string",
743
+ "enum": [
744
+ "first_block",
745
+ "faster",
746
+ "mag",
747
+ "taylorseer",
748
+ "text_kv"
749
+ ]
750
+ },
751
+ "threshold": {
752
+ "description": "Cache threshold for first_block and mag types. Higher = more speedup, more quality loss. first_block default: 0.05, mag default: 0.06. Accepts a 'variable:' reference.",
753
+ "type": [
754
+ "number",
755
+ "string"
756
+ ]
757
+ },
758
+ "num_inference_steps": {
759
+ "description": "Required for mag cache type. Must match the pipeline's num_inference_steps.",
760
+ "type": "integer"
761
+ },
762
+ "max_skip_steps": {
763
+ "description": "Max consecutive skippable steps for mag cache. Default: 3.",
764
+ "type": "integer"
765
+ },
766
+ "retention_ratio": {
767
+ "description": "Fraction of initial steps where skipping is disabled for mag cache. Default: 0.2.",
768
+ "type": "number"
769
+ },
770
+ "mag_ratios": {
771
+ "description": "Per-step magnitude ratios for mag cache. Required unless calibrate is true, and checkpoint-dependent. Either a preset name shipped by diffusers ('flux') or an explicit array of per-step ratios. Interpolated automatically when its length differs from num_inference_steps.",
772
+ "oneOf": [
773
+ {
774
+ "type": "string"
775
+ },
776
+ {
777
+ "type": "array",
778
+ "items": {
779
+ "type": "number"
780
+ }
781
+ }
782
+ ]
783
+ },
784
+ "calibrate": {
785
+ "description": "Run mag cache in calibration mode: skip nothing, and log the magnitude ratios for this model at the end of the run so they can be pasted into mag_ratios. Default: false.",
786
+ "type": "boolean"
787
+ },
788
+ "cache_interval": {
789
+ "description": "Full computation every N steps for taylorseer cache. Default: 5.",
790
+ "type": "integer"
791
+ },
792
+ "max_order": {
793
+ "description": "Taylor series order for taylorseer cache. Higher = better approximation, more memory. Default: 1.",
794
+ "type": "integer"
795
+ }
796
+ },
797
+ "required": [
798
+ "type"
799
+ ]
800
+ },
801
+ "teacache": {
802
+ "description": "Enable TeaCache inference acceleration. Auto-detects transformer type. Caches intermediate computations and skips redundant steps. Requires num_inference_steps in pipeline arguments. Currently supports FluxTransformer2DModel. Mutually exclusive with cache.",
803
+ "type": "object",
804
+ "properties": {
805
+ "rel_l1_thresh": {
806
+ "description": "Cache threshold. Higher = more speedup, more quality loss. Model-specific defaults apply if omitted.",
807
+ "type": "number"
808
+ },
809
+ "coefficients": {
810
+ "description": "Polynomial coefficients for rescaling relative L1 distance. 5 floats for np.poly1d. If omitted, uses model-specific defaults.",
811
+ "type": "array",
812
+ "items": {
813
+ "type": "number"
814
+ },
815
+ "minItems": 5,
816
+ "maxItems": 5
817
+ },
818
+ "variant": {
819
+ "description": "Explicit model variant from teacache_models.json. Required for models with multiple variants (e.g., 'cogvideox_2b', 'wan2.1_t2v_1.3b'). If omitted, uses the class default.",
820
+ "type": "string"
821
+ }
822
+ }
823
+ },
824
+ "vae": {
825
+ "type": "object",
826
+ "properties": {
827
+ "enable_slicing": {
828
+ "description": "Decode the latent batch one sample at a time rather than all at once - the answer to a decode that runs out of memory on a batch the denoise itself fitted. Costs nothing but the loss of batched decode parallelism.",
829
+ "type": "boolean"
830
+ },
831
+ "enable_tiling": {
832
+ "description": "Decode in tiles rather than in one allocation, for a single sample too large to decode whole - a long or high-resolution video, where the peak is the decode rather than the denoise. Combines with 'enable_slicing'.",
833
+ "type": "boolean"
834
+ },
835
+ "channels_last": {
836
+ "type": "boolean"
837
+ },
838
+ "torch_dtype": {
839
+ "type": "string"
840
+ }
841
+ }
842
+ },
843
+ "unet": {
844
+ "type": "object",
845
+ "properties": {
846
+ "enable_forward_chunking": {
847
+ "type": "boolean"
848
+ },
849
+ "channels_last": {
850
+ "type": "boolean"
851
+ },
852
+ "torch_dtype": {
853
+ "type": "string"
854
+ },
855
+ "attn_processor_type": {
856
+ "type": "string"
857
+ }
858
+ }
859
+ },
860
+ "transformer": {
861
+ "type": "object",
862
+ "properties": {
863
+ "attn_processor_type": {
864
+ "description": "The type name of an attention processor",
865
+ "type": "string"
866
+ }
867
+ }
868
+ },
869
+ "text_encoder": {
870
+ "type": "object",
871
+ "properties": {
872
+ "torch_dtype": {
873
+ "type": "string"
874
+ }
875
+ }
876
+ },
877
+ "shared_components": {
878
+ "$ref": "#/$defs/shared_components"
879
+ },
880
+ "reused_components": {
881
+ "$ref": "#/$defs/reused_components"
882
+ },
883
+ "enable_layerwise_casting": {
884
+ "$ref": "#/$defs/enable_layerwise_casting"
885
+ },
886
+ "inversion": {
887
+ "description": "Run the pipeline's invert() rather than the pipeline itself, returning the inverted latents. Only for pipelines that have one.",
888
+ "type": "boolean"
889
+ },
890
+ "generate": {
891
+ "description": "Run the pipeline's generate() rather than the pipeline itself, returning the generated ids. Only for pipelines that have one, such as a conditioner used on its own.",
892
+ "type": "boolean"
893
+ }
894
+ },
895
+ "additionalProperties": false,
896
+ "required": [
897
+ "component_type"
898
+ ]
899
+ },
900
+ "shared_components": {
901
+ "description": "The names of components this step loads and later steps may reuse. A modular pipeline is given them after it is built, so the components it shares are the ones it actually loaded.",
902
+ "type": "array",
903
+ "items": {
904
+ "type": "string"
905
+ }
906
+ },
907
+ "reused_components": {
908
+ "description": "The names of components an earlier step shared, reused here instead of being loaded again. A reused component keeps the device placement the step that shared it gave it.",
909
+ "type": "array",
910
+ "items": {
911
+ "type": "string"
912
+ }
913
+ },
914
+ "scheduler": {
915
+ "type": "object",
916
+ "properties": {
917
+ "configuration": {
918
+ "type": "object",
919
+ "properties": {
920
+ "scheduler_type": {
921
+ "type": "string"
922
+ }
923
+ },
924
+ "required": [
925
+ "scheduler_type"
926
+ ]
927
+ },
928
+ "from_config_args": {
929
+ "$ref": "#/$defs/arguments"
930
+ },
931
+ "shift": {
932
+ "description": "Exponential sigma shift applied to this scheduler's schedule, for schedulers that take one (MiniMax H3: 12.0 for video, 3.0 for audio in the released checkpoint). Lower it when running few steps - a high shift packs the whole grid near sigma 1 and leaves one enormous final step, which a short schedule cannot absorb. Accepts a 'variable:' reference.",
933
+ "type": [
934
+ "number",
935
+ "string"
936
+ ],
937
+ "exclusiveMinimum": 0
938
+ }
939
+ },
940
+ "anyOf": [
941
+ {
942
+ "required": [
943
+ "configuration"
944
+ ]
945
+ },
946
+ {
947
+ "required": [
948
+ "shift"
949
+ ]
950
+ }
951
+ ]
952
+ },
953
+ "pipeline_component": {
954
+ "type": "object",
955
+ "properties": {
956
+ "configuration": {
957
+ "type": "object",
958
+ "properties": {
959
+ "device": {
960
+ "description": "The device to load this component onto, e.g. 'cuda', 'cuda:1', 'mps' or 'cpu'. Defaults to the device its pipeline runs on. A backend this machine does not have is translated to the one it does, with a warning.",
961
+ "type": "string"
962
+ },
963
+ "component_type": {
964
+ "description": "The python type of pipeline component to use in the format 'module.typename' module defaults to diffusers",
965
+ "type": "string"
966
+ }
967
+ },
968
+ "required": [
969
+ "component_type"
970
+ ]
971
+ },
972
+ "group_offload": {
973
+ "$ref": "#/$defs/group_offload"
974
+ },
975
+ "enable_layerwise_casting": {
976
+ "$ref": "#/$defs/enable_layerwise_casting"
977
+ },
978
+ "quantization_config": {
979
+ "$ref": "#/$defs/quantization_config"
980
+ },
981
+ "from_pretrained_arguments": {
982
+ "$ref": "#/$defs/from_pretrained_arguments"
983
+ }
984
+ },
985
+ "required": [
986
+ "configuration",
987
+ "from_pretrained_arguments"
988
+ ]
989
+ },
990
+ "quantization_config": {
991
+ "type": "object",
992
+ "properties": {
993
+ "configuration": {
994
+ "type": "object",
995
+ "properties": {
996
+ "config_type": {
997
+ "type": "string"
998
+ }
999
+ },
1000
+ "required": [
1001
+ "config_type"
1002
+ ]
1003
+ },
1004
+ "arguments": {
1005
+ "$ref": "#/$defs/arguments"
1006
+ }
1007
+ },
1008
+ "required": [
1009
+ "configuration",
1010
+ "arguments"
1011
+ ]
1012
+ },
1013
+ "controlnet": {
1014
+ "type": "object",
1015
+ "properties": {
1016
+ "configuration": {
1017
+ "$ref": "#/$defs/pipeline_configuration"
1018
+ },
1019
+ "from_pretrained_arguments": {
1020
+ "$ref": "#/$defs/from_pretrained_arguments"
1021
+ }
1022
+ },
1023
+ "required": [
1024
+ "configuration",
1025
+ "from_pretrained_arguments"
1026
+ ]
1027
+ },
1028
+ "ip_adapter": {
1029
+ "type": "object",
1030
+ "properties": {
1031
+ "model_name": {
1032
+ "description": "The huggingface hub name of the ip adadapter model to use.",
1033
+ "type": "string"
1034
+ },
1035
+ "weight_name": {
1036
+ "description": "The file name of the ip adapter weights.",
1037
+ "type": "string"
1038
+ },
1039
+ "subfolder": {
1040
+ "description": "The subfolder location of a model file within a larger model repository",
1041
+ "type": [
1042
+ "string",
1043
+ "null"
1044
+ ]
1045
+ },
1046
+ "scale": {
1047
+ "description": "The scale factor of the ip adapter.",
1048
+ "type": "number",
1049
+ "format": "float"
1050
+ }
1051
+ },
1052
+ "required": [
1053
+ "model_name"
1054
+ ]
1055
+ },
1056
+ "lora": {
1057
+ "type": "object",
1058
+ "properties": {
1059
+ "model_name": {
1060
+ "description": "The huggingface hub name of the lora.",
1061
+ "type": "string"
1062
+ },
1063
+ "weight_name": {
1064
+ "description": "The file name of the lora weights.",
1065
+ "type": "string"
1066
+ },
1067
+ "subfolder": {
1068
+ "description": "The subfolder location of a model file within a larger model repository",
1069
+ "type": "string"
1070
+ },
1071
+ "scale": {
1072
+ "description": "The scale factor of the lora when fusing. Accepts a 'variable:' reference.",
1073
+ "type": [
1074
+ "number",
1075
+ "string"
1076
+ ],
1077
+ "format": "float"
1078
+ },
1079
+ "alpha": {
1080
+ "description": "The network alpha to scale this lora by, overriding the one its checkpoint declares. peft scales an adapter by 'scale * alpha / rank', taking alpha from the file - a per-module '.alpha' tensor, else a '__metadata__' alpha where the loader honors one, else the rank itself. State it where the checkpoint's own figure is not the one it is meant to run at: the 768p MiniMax-H3 turbo LoRAs record alpha 8 at rank 128 while upstream runs them at 128. Accepts a 'variable:' reference.",
1081
+ "type": [
1082
+ "number",
1083
+ "string",
1084
+ "null"
1085
+ ],
1086
+ "exclusiveMinimum": 0
1087
+ },
1088
+ "adapter_name": {
1089
+ "description": "The user defined name of the adapter to pass to set_adapters function",
1090
+ "type": "string"
1091
+ }
1092
+ },
1093
+ "required": [
1094
+ "model_name"
1095
+ ]
1096
+ },
1097
+ "from_pretrained_arguments": {
1098
+ "type": "object",
1099
+ "description": "Arguments to pass to the from_pretrained function when creating the component. Names at most one source: 'model_name', 'from_single_file' or 'task'. A pipeline assembled entirely from components an earlier step shared and components declared beside it - LTX-2's latent upsampler, say - names none of them and is constructed directly from those components.",
1100
+ "properties": {
1101
+ "model_name": {
1102
+ "description": "The huggingface hub name of the pretrained model to load",
1103
+ "type": "string"
1104
+ },
1105
+ "from_single_file": {
1106
+ "description": "Location of a checkpoint model from a single file",
1107
+ "type": "string"
1108
+ },
1109
+ "task": {
1110
+ "description": "The name of a task that transformers.Pipeline understands",
1111
+ "type": "string"
1112
+ }
1113
+ },
1114
+ "additionalProperties": {
1115
+ "type": [
1116
+ "string",
1117
+ "number",
1118
+ "object",
1119
+ "array",
1120
+ "boolean",
1121
+ "null"
1122
+ ]
1123
+ },
1124
+ "oneOf": [
1125
+ {
1126
+ "required": [
1127
+ "model_name"
1128
+ ]
1129
+ },
1130
+ {
1131
+ "required": [
1132
+ "from_single_file"
1133
+ ]
1134
+ },
1135
+ {
1136
+ "required": [
1137
+ "task"
1138
+ ]
1139
+ },
1140
+ {
1141
+ "allOf": [
1142
+ {
1143
+ "not": {
1144
+ "required": [
1145
+ "model_name"
1146
+ ]
1147
+ }
1148
+ },
1149
+ {
1150
+ "not": {
1151
+ "required": [
1152
+ "from_single_file"
1153
+ ]
1154
+ }
1155
+ },
1156
+ {
1157
+ "not": {
1158
+ "required": [
1159
+ "task"
1160
+ ]
1161
+ }
1162
+ }
1163
+ ]
1164
+ }
1165
+ ]
1166
+ },
1167
+ "task": {
1168
+ "type": "object",
1169
+ "additionalProperties": false,
1170
+ "$comment": "Closed - see 'step'. A task's own arguments are open; it is the task object itself that is fixed.",
1171
+ "properties": {
1172
+ "command": {
1173
+ "type": "string"
1174
+ },
1175
+ "arguments": {
1176
+ "$ref": "#/$defs/arguments"
1177
+ },
1178
+ "inputs": {
1179
+ "type": "array",
1180
+ "items": {
1181
+ "type": [
1182
+ "string",
1183
+ "integer",
1184
+ "number",
1185
+ "object",
1186
+ "boolean"
1187
+ ]
1188
+ }
1189
+ }
1190
+ },
1191
+ "oneOf": [
1192
+ {
1193
+ "required": [
1194
+ "command",
1195
+ "arguments"
1196
+ ]
1197
+ },
1198
+ {
1199
+ "required": [
1200
+ "command",
1201
+ "inputs"
1202
+ ]
1203
+ }
1204
+ ]
1205
+ },
1206
+ "workflow_reference": {
1207
+ "type": "object",
1208
+ "additionalProperties": false,
1209
+ "$comment": "Closed - see 'step'.",
1210
+ "properties": {
1211
+ "path": {
1212
+ "description": "The workflow this step composes: a catalog name as list_workflows reports it (with or without '.json'), a path relative to the workflow that names it ('../models/x.json'), or 'builtin:name.json' for one of the packaged fragments. A name is resolved first beside the referencing file, then against this run's own workflows directory, then against each read-only source on the server's search path - so a stored template can be composed without copying it into the workspace. A path outside every source is refused. When the composing step declares a 'result', that is where the composed output is saved and the composed workflow's own last step does not save it again.",
1213
+ "type": "string"
1214
+ },
1215
+ "arguments": {
1216
+ "$ref": "#/$defs/arguments"
1217
+ }
1218
+ }
1219
+ },
1220
+ "result": {
1221
+ "type": "object",
1222
+ "properties": {
1223
+ "content_type": {
1224
+ "description": "The content type of the result when serialized to disk. Audio can be written as 'audio/wav', 'audio/flac', 'audio/mpeg' (mp3), 'audio/ogg' or 'audio/opus'. Opus is written into an ogg container and only encodes sample rates of 8000, 12000, 16000, 24000 or 48000.",
1225
+ "type": "string"
1226
+ },
1227
+ "save": {
1228
+ "description": "Whether to save the result",
1229
+ "type": "boolean",
1230
+ "default": true
1231
+ },
1232
+ "file_base_name": {
1233
+ "description": "The base name for saving result and metadata files. It replaces the name the engine would derive from the workflow and step - it is not a prefix on one - so a caller that sets it can predict the file name. That derived name is what makes two steps' files distinct, so give each step that sets one a different value, or the second collides and gets a '-2' counter. A name, not a path - it may not contain a separator; to place a step's files in a subfolder of the run directory set 'subfolder'. The file extension is determined by context and content_type.",
1234
+ "type": "string"
1235
+ },
1236
+ "subfolder": {
1237
+ "description": "A subfolder of the run directory this step's files are written into, as a relative path like 'final' or 'shots/act-1'. Optional: without it the files land at the root of the run directory as they always have. By convention a step whose output the user will be shown is 'final' and everything else is 'intermediate', which is how a consumer tells the deliverable from the scratch files without knowing the workflow; the engine treats no name specially. May be a 'variable:' or, inside a for_each step, an 'item:' reference. Each segment must start with a letter, digit or underscore; '..' and '\\' are refused.",
1238
+ "type": "string"
1239
+ },
1240
+ "fps": {
1241
+ "description": "Frames per second - only used when output is video. Defaults to the rate the frames themselves carry (a chain's 'fps', a join's own rate, the rate of the file a task read) and to 8 only when nothing knows better. Set it to write at a rate other than the source's - a deliberate slow motion - which the run then warns about. May be a 'variable:' or, inside a for_each step, an 'item:' reference, resolved the same way 'subfolder' is - the resolved value must be a positive whole number; a fractional rate is refused, since the video writer would silently truncate rather than honor it.",
1242
+ "type": ["integer", "number", "string"],
1243
+ "default": 8
1244
+ },
1245
+ "sample_rate": {
1246
+ "description": "Audio sample rate - only used when output is audio",
1247
+ "type": "integer",
1248
+ "default": 44100
1249
+ },
1250
+ "audio_sample_rate": {
1251
+ "description": "Sample rate of the audio generated with a video, muxed into 'video/mp4' output. Defaults to the rate reported by the pipeline's vocoder - only set this to override it.",
1252
+ "type": "integer"
1253
+ },
1254
+ "subtype": {
1255
+ "description": "Audio encoding subtype, e.g. 'PCM_24' for wav and flac or 'VORBIS' for ogg. Defaults to the container's own default ('PCM_16' for wav and flac). Only used when output is audio.",
1256
+ "type": "string"
1257
+ },
1258
+ "compression_level": {
1259
+ "description": "Encoding compression level from 0.0 to 1.0 for compressed audio formats (flac, mp3, ogg). Higher means smaller files. Only used when output is audio.",
1260
+ "type": "number",
1261
+ "minimum": 0,
1262
+ "maximum": 1
1263
+ },
1264
+ "bitrate_mode": {
1265
+ "description": "Bitrate mode for compressed audio formats: 'CONSTANT', 'AVERAGE' or 'VARIABLE'. Only used when output is audio.",
1266
+ "type": "string",
1267
+ "enum": [
1268
+ "CONSTANT",
1269
+ "AVERAGE",
1270
+ "VARIABLE"
1271
+ ]
1272
+ },
1273
+ "embed_metadata": {
1274
+ "description": "Whether to embed generation parameters as metadata in saved images (PNG info chunks or EXIF). Only applies to image content types.",
1275
+ "type": "boolean",
1276
+ "default": false
1277
+ },
1278
+ "format": {
1279
+ "description": "Audio container format handed to the writer, e.g. 'WAV' or 'FLAC'. Derived from content_type unless set. Only used when output is audio.",
1280
+ "type": "string"
1281
+ }
1282
+ },
1283
+ "additionalProperties": false,
1284
+ "required": [
1285
+ "content_type"
1286
+ ]
1287
+ },
1288
+ "enable_layerwise_casting": {
1289
+ "type": "object",
1290
+ "properties": {
1291
+ "storage_dtype": {
1292
+ "type": "string"
1293
+ },
1294
+ "compute_dtype": {
1295
+ "type": "string"
1296
+ }
1297
+ },
1298
+ "required": [
1299
+ "storage_dtype",
1300
+ "compute_dtype"
1301
+ ]
1302
+ },
1303
+ "group_offload": {
1304
+ "description": "Stream a model between system memory and the accelerator a group of layers at a time. Anything beyond the keys named here is passed through to apply_group_offloading, e.g. 'use_stream', 'num_blocks_per_group', 'low_cpu_mem_usage', 'offload_to_disk_path'.",
1305
+ "type": "object",
1306
+ "properties": {
1307
+ "onload_device": {
1308
+ "type": "string"
1309
+ },
1310
+ "offload_device": {
1311
+ "type": "string",
1312
+ "default": "cpu"
1313
+ },
1314
+ "offload_type": {
1315
+ "type": "string"
1316
+ }
1317
+ },
1318
+ "additionalProperties": true,
1319
+ "required": [
1320
+ "offload_type"
1321
+ ]
1322
+ },
1323
+ "compile_config": {
1324
+ "description": "Compile the component with torch.compile once it is fully configured. Best paired with a pinned attention_backend - the per-call context manager forces recompiles.",
1325
+ "type": "object",
1326
+ "properties": {
1327
+ "mode": {
1328
+ "description": "torch.compile mode, e.g. 'default', 'reduce-overhead', 'max-autotune'.",
1329
+ "type": "string"
1330
+ },
1331
+ "fullgraph": {
1332
+ "description": "Require a single graph with no breaks. Fails fast on graph breaks instead of silently losing speedup.",
1333
+ "type": "boolean"
1334
+ },
1335
+ "dynamic": {
1336
+ "description": "Compile with dynamic shapes. Set true when resolutions or frame counts vary between runs to avoid recompiles.",
1337
+ "type": "boolean"
1338
+ },
1339
+ "repeated_blocks": {
1340
+ "description": "Compile only the model's repeated block classes (diffusers regional compilation) - near the same speedup as full compilation with a fraction of the cold-start cost.",
1341
+ "type": "boolean"
1342
+ }
1343
+ }
1344
+ }
1345
+ }
1346
+ }