diffusers-workflow 0.4.0a3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0a3.dist-info/METADATA +310 -0
- diffusers_workflow-0.4.0a3.dist-info/RECORD +171 -0
- diffusers_workflow-0.4.0a3.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0a3.dist-info/entry_points.txt +6 -0
- diffusers_workflow-0.4.0a3.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0a3.dist-info/top_level.txt +1 -0
- dw/__init__.py +353 -0
- dw/arguments.py +906 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +145 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/events.py +78 -0
- dw/hub_cache.py +289 -0
- dw/introspection.py +458 -0
- dw/log_setup.py +45 -0
- dw/pipeline_processors/chain.py +750 -0
- dw/pipeline_processors/config_objects.py +235 -0
- dw/pipeline_processors/pipeline.py +1687 -0
- dw/pipeline_processors/remote.py +18 -0
- dw/previous_results.py +259 -0
- dw/prompt_weighting.py +378 -0
- dw/repl.py +298 -0
- dw/repl_commands.py +808 -0
- dw/repl_worker.py +129 -0
- dw/result.py +850 -0
- dw/run.py +92 -0
- dw/schema.py +24 -0
- dw/security.py +379 -0
- dw/serve.py +70 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +588 -0
- dw/server/jobs.py +547 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CEh6hWi2.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CExg3_mM.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-DH6orYh2.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-CbrMVW4Q.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-YDNPZw2M.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-B_zSGWO2.js +1 -0
- dw/server/ui/assets/index-B7-VcYS-.css +1 -0
- dw/server/ui/assets/index-D_EiPU3b.js +13 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-dYuBvioq.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-CUqLM39V.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-D6vxBzMv.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-1WJ2palX.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-SHQb6vmD.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-CP-s5rcP.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-x0_EGHq9.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-BZC4LQDP.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-BTfA6SbD.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-CWA4MsNk.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-CWU0uvj5.js +1 -0
- dw/server/ui/assets/xml-KmfTm3rg.js +1 -0
- dw/server/ui/assets/yaml-nFO_dDS6.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/settings.py +77 -0
- dw/step.py +132 -0
- dw/tasks/audio_utils.py +266 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/concat_videos.py +80 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +139 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +661 -0
- dw/tasks/interpolate_frames.py +227 -0
- dw/tasks/model_cache.py +39 -0
- dw/tasks/pair_audio.py +58 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/task.py +474 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +168 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +154 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +376 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +68 -0
- dw/validate.py +43 -0
- dw/variables.py +153 -0
- dw/worker.py +517 -0
- dw/workflow.py +553 -0
- dw/workflow_schema.json +1157 -0
- dw/workflows/augment_prompt.json +65 -0
- dw/workflows/describe_image.json +58 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
dw/workflow_schema.json
ADDED
|
@@ -0,0 +1,1157 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$id": "https://github.com/dkackman/diffusers-helper/workflow",
|
|
3
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema#",
|
|
4
|
+
"description": "The definition of the diffusers-workflow.",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"properties": {
|
|
7
|
+
"id": {
|
|
8
|
+
"type": "string"
|
|
9
|
+
},
|
|
10
|
+
"description": {
|
|
11
|
+
"type": "string"
|
|
12
|
+
},
|
|
13
|
+
"variables": {
|
|
14
|
+
"$ref": "#/$defs/arguments"
|
|
15
|
+
},
|
|
16
|
+
"seed": {
|
|
17
|
+
"description": "Default seed for the entire workflow",
|
|
18
|
+
"type": "integer",
|
|
19
|
+
"format": "int64"
|
|
20
|
+
},
|
|
21
|
+
"steps": {
|
|
22
|
+
"type": "array",
|
|
23
|
+
"minItems": 1,
|
|
24
|
+
"items": {
|
|
25
|
+
"$ref": "#/$defs/step"
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"required": [
|
|
30
|
+
"id",
|
|
31
|
+
"steps"
|
|
32
|
+
],
|
|
33
|
+
"$defs": {
|
|
34
|
+
"image": {
|
|
35
|
+
"type": "object",
|
|
36
|
+
"properties": {
|
|
37
|
+
"location": {
|
|
38
|
+
"type": "string",
|
|
39
|
+
"format": "uri"
|
|
40
|
+
},
|
|
41
|
+
"size": {
|
|
42
|
+
"type": "object",
|
|
43
|
+
"properties": {
|
|
44
|
+
"width": {
|
|
45
|
+
"type": "integer",
|
|
46
|
+
"format": "int16"
|
|
47
|
+
},
|
|
48
|
+
"height": {
|
|
49
|
+
"type": "integer",
|
|
50
|
+
"format": "int16"
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
},
|
|
55
|
+
"required": [
|
|
56
|
+
"location"
|
|
57
|
+
]
|
|
58
|
+
},
|
|
59
|
+
"step": {
|
|
60
|
+
"type": "object",
|
|
61
|
+
"properties": {
|
|
62
|
+
"name": {
|
|
63
|
+
"type": "string"
|
|
64
|
+
},
|
|
65
|
+
"seed": {
|
|
66
|
+
"description": "Default seed for the entire step",
|
|
67
|
+
"type": "integer",
|
|
68
|
+
"format": "int64"
|
|
69
|
+
},
|
|
70
|
+
"release_pipeline": {
|
|
71
|
+
"description": "Unload this step's pipeline once the step completes, freeing its memory for later steps. A later pipeline_reference to this step is an error, and the REPL's cross-run cache will not retain it.",
|
|
72
|
+
"type": "boolean"
|
|
73
|
+
},
|
|
74
|
+
"release_models": {
|
|
75
|
+
"description": "Unload every cached task model once the step completes, freeing their memory for later steps. Task models otherwise stay loaded for the life of the process; set this on a task or sub-workflow step whose model is not needed again, such as a prompt expander running ahead of a generation step. A later step using the same model reloads it.",
|
|
76
|
+
"type": "boolean"
|
|
77
|
+
},
|
|
78
|
+
"task": {
|
|
79
|
+
"$ref": "#/$defs/task"
|
|
80
|
+
},
|
|
81
|
+
"pipeline": {
|
|
82
|
+
"$ref": "#/$defs/pipeline"
|
|
83
|
+
},
|
|
84
|
+
"pipeline_reference": {
|
|
85
|
+
"$ref": "#/$defs/pipeline_reference"
|
|
86
|
+
},
|
|
87
|
+
"workflow": {
|
|
88
|
+
"$ref": "#/$defs/workflow_reference"
|
|
89
|
+
},
|
|
90
|
+
"result": {
|
|
91
|
+
"$ref": "#/$defs/result"
|
|
92
|
+
}
|
|
93
|
+
},
|
|
94
|
+
"oneOf": [
|
|
95
|
+
{
|
|
96
|
+
"required": [
|
|
97
|
+
"name",
|
|
98
|
+
"task"
|
|
99
|
+
]
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
"required": [
|
|
103
|
+
"name",
|
|
104
|
+
"pipeline"
|
|
105
|
+
]
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
"required": [
|
|
109
|
+
"name",
|
|
110
|
+
"pipeline_reference"
|
|
111
|
+
]
|
|
112
|
+
},
|
|
113
|
+
{
|
|
114
|
+
"required": [
|
|
115
|
+
"name",
|
|
116
|
+
"workflow"
|
|
117
|
+
]
|
|
118
|
+
}
|
|
119
|
+
]
|
|
120
|
+
},
|
|
121
|
+
"arguments": {
|
|
122
|
+
"type": "object",
|
|
123
|
+
"additionalProperties": {
|
|
124
|
+
"description": "null declares an argument that is optional - a workflow can expose a variable a caller may pass without inventing a sentinel value for its absence.",
|
|
125
|
+
"type": [
|
|
126
|
+
"string",
|
|
127
|
+
"integer",
|
|
128
|
+
"number",
|
|
129
|
+
"object",
|
|
130
|
+
"array",
|
|
131
|
+
"boolean",
|
|
132
|
+
"null"
|
|
133
|
+
]
|
|
134
|
+
}
|
|
135
|
+
},
|
|
136
|
+
"pipeline_reference": {
|
|
137
|
+
"type": "object",
|
|
138
|
+
"properties": {
|
|
139
|
+
"reference_name": {
|
|
140
|
+
"type": "string"
|
|
141
|
+
},
|
|
142
|
+
"chain": {
|
|
143
|
+
"$ref": "#/$defs/chain"
|
|
144
|
+
},
|
|
145
|
+
"arguments": {
|
|
146
|
+
"$ref": "#/$defs/arguments"
|
|
147
|
+
}
|
|
148
|
+
},
|
|
149
|
+
"required": [
|
|
150
|
+
"reference_name",
|
|
151
|
+
"arguments"
|
|
152
|
+
]
|
|
153
|
+
},
|
|
154
|
+
"chain": {
|
|
155
|
+
"description": "Run this pipeline repeatedly, carrying visual continuity from each segment into the next, and stitch the segments into one long video. Specify the length with exactly one of 'segments' or 'match_audio'.",
|
|
156
|
+
"type": "object",
|
|
157
|
+
"properties": {
|
|
158
|
+
"segments": {
|
|
159
|
+
"description": "Number of segments to generate and stitch. Accepts a 'variable:' reference.",
|
|
160
|
+
"type": [
|
|
161
|
+
"integer",
|
|
162
|
+
"string"
|
|
163
|
+
],
|
|
164
|
+
"minimum": 1
|
|
165
|
+
},
|
|
166
|
+
"match_audio": {
|
|
167
|
+
"description": "Derive the segment count from the duration of the audio reference in this step's arguments, slicing it into frame-aligned per-segment chunks; the final video is muxed with the original, unsliced audio track.",
|
|
168
|
+
"type": "boolean"
|
|
169
|
+
},
|
|
170
|
+
"continuity": {
|
|
171
|
+
"description": "How continuity is carried from one segment into the next. 'last_frame' carries the segment's final frame as the next one's keyframe. 'last_segment' carries the segment itself - its frames and the soundtrack generated with them - as a video reference, which keeps motion, camera and voice across the seam instead of appearance alone; it needs a 'segment_argument' holding a references list.",
|
|
172
|
+
"type": "string",
|
|
173
|
+
"enum": [
|
|
174
|
+
"last_frame",
|
|
175
|
+
"last_segment"
|
|
176
|
+
],
|
|
177
|
+
"default": "last_frame"
|
|
178
|
+
},
|
|
179
|
+
"segment_argument": {
|
|
180
|
+
"description": "The pipeline argument the carry-over lands in - e.g. 'image' for image-to-video pipelines, or 'references' for reference-conditioned modular pipelines, where the carry-over is appended as an image or video reference.",
|
|
181
|
+
"type": "string",
|
|
182
|
+
"default": "image"
|
|
183
|
+
},
|
|
184
|
+
"carry_frames": {
|
|
185
|
+
"description": "'last_segment' only - carry just this many frames from the tail of each segment rather than all of them, cutting the reference's soundtrack to the same span. A shorter carry costs less sequence length and memory (MiniMax H3 conditions on reference clips of 2 seconds and up, i.e. 48 frames at its 24 fps). Accepts a 'variable:' reference.",
|
|
186
|
+
"type": [
|
|
187
|
+
"integer",
|
|
188
|
+
"string"
|
|
189
|
+
],
|
|
190
|
+
"minimum": 1
|
|
191
|
+
},
|
|
192
|
+
"carry_audio": {
|
|
193
|
+
"description": "'last_segment' only - whether the carried video reference brings the soundtrack generated with it, which is what carries a voice across the seam. Turn it off when the segments are already conditioned on a supplied audio track, as in a 'match_audio' chain.",
|
|
194
|
+
"type": "boolean",
|
|
195
|
+
"default": true
|
|
196
|
+
},
|
|
197
|
+
"trim_frames": {
|
|
198
|
+
"description": "Frames removed from the head of every segment after the first (an image-to-video segment reproduces its keyframe as frame 0). Also bounds the audio crossfade window to trim_frames / fps seconds. Accepts a 'variable:' reference.",
|
|
199
|
+
"type": [
|
|
200
|
+
"integer",
|
|
201
|
+
"string"
|
|
202
|
+
],
|
|
203
|
+
"minimum": 0,
|
|
204
|
+
"default": 1
|
|
205
|
+
},
|
|
206
|
+
"crossfade_ms": {
|
|
207
|
+
"description": "Equal-power crossfade applied to generated audio at segment boundaries, clamped to the trim_frames / fps window. Accepts a 'variable:' reference.",
|
|
208
|
+
"type": [
|
|
209
|
+
"number",
|
|
210
|
+
"string"
|
|
211
|
+
],
|
|
212
|
+
"minimum": 0,
|
|
213
|
+
"default": 75
|
|
214
|
+
},
|
|
215
|
+
"fps": {
|
|
216
|
+
"description": "Frame rate used for the chain's audio math - trimming, crossfading, and match_audio planning. Defaults to the pipeline's 'frame_rate' argument when present; pipelines with a fixed rate (MiniMax H3: 24) need it set here. Accepts a 'variable:' reference.",
|
|
217
|
+
"type": [
|
|
218
|
+
"number",
|
|
219
|
+
"string"
|
|
220
|
+
],
|
|
221
|
+
"exclusiveMinimum": 0
|
|
222
|
+
},
|
|
223
|
+
"frame_snap": {
|
|
224
|
+
"description": "Constraint the pipeline places on num_frames - counts must equal modulus * n + remainder within the bounds (MiniMax H3: modulus 17, remainder 5, min 124, max 345). Used to snap the final match_audio segment to a valid length.",
|
|
225
|
+
"type": "object",
|
|
226
|
+
"properties": {
|
|
227
|
+
"modulus": {
|
|
228
|
+
"type": "integer",
|
|
229
|
+
"minimum": 1
|
|
230
|
+
},
|
|
231
|
+
"remainder": {
|
|
232
|
+
"type": "integer",
|
|
233
|
+
"minimum": 0
|
|
234
|
+
},
|
|
235
|
+
"min_frames": {
|
|
236
|
+
"type": "integer",
|
|
237
|
+
"minimum": 1
|
|
238
|
+
},
|
|
239
|
+
"max_frames": {
|
|
240
|
+
"type": "integer",
|
|
241
|
+
"minimum": 1
|
|
242
|
+
}
|
|
243
|
+
},
|
|
244
|
+
"required": [
|
|
245
|
+
"modulus",
|
|
246
|
+
"remainder"
|
|
247
|
+
],
|
|
248
|
+
"additionalProperties": false
|
|
249
|
+
},
|
|
250
|
+
"prompts": {
|
|
251
|
+
"description": "Per-segment prompt overrides; segment i uses prompts[min(i, len(prompts) - 1)].",
|
|
252
|
+
"type": "array",
|
|
253
|
+
"items": {
|
|
254
|
+
"type": "string"
|
|
255
|
+
},
|
|
256
|
+
"minItems": 1
|
|
257
|
+
},
|
|
258
|
+
"save_segments": {
|
|
259
|
+
"description": "Write each completed segment to the output directory as a playable mp4 and free its frames, bounding memory to one segment - a crashed chain leaves the finished segments behind. The final video is streamed from the segment files. Requires PyAV and a frame rate.",
|
|
260
|
+
"type": "boolean",
|
|
261
|
+
"default": false
|
|
262
|
+
},
|
|
263
|
+
"keep_segments": {
|
|
264
|
+
"description": "Leave the segment files in place after the final video is written (default: they are removed once it saves successfully). Only meaningful with save_segments.",
|
|
265
|
+
"type": "boolean",
|
|
266
|
+
"default": false
|
|
267
|
+
}
|
|
268
|
+
},
|
|
269
|
+
"oneOf": [
|
|
270
|
+
{
|
|
271
|
+
"required": [
|
|
272
|
+
"segments"
|
|
273
|
+
]
|
|
274
|
+
},
|
|
275
|
+
{
|
|
276
|
+
"required": [
|
|
277
|
+
"match_audio"
|
|
278
|
+
]
|
|
279
|
+
}
|
|
280
|
+
],
|
|
281
|
+
"additionalProperties": false
|
|
282
|
+
},
|
|
283
|
+
"pipeline": {
|
|
284
|
+
"type": "object",
|
|
285
|
+
"properties": {
|
|
286
|
+
"configuration": {
|
|
287
|
+
"$ref": "#/$defs/pipeline_configuration"
|
|
288
|
+
},
|
|
289
|
+
"shared_components": {
|
|
290
|
+
"$ref": "#/$defs/shared_components"
|
|
291
|
+
},
|
|
292
|
+
"reused_components": {
|
|
293
|
+
"$ref": "#/$defs/reused_components"
|
|
294
|
+
},
|
|
295
|
+
"scheduler": {
|
|
296
|
+
"$ref": "#/$defs/scheduler"
|
|
297
|
+
},
|
|
298
|
+
"audio_scheduler": {
|
|
299
|
+
"$ref": "#/$defs/scheduler"
|
|
300
|
+
},
|
|
301
|
+
"model": {
|
|
302
|
+
"$ref": "#/$defs/pipeline_component"
|
|
303
|
+
},
|
|
304
|
+
"transformer": {
|
|
305
|
+
"$ref": "#/$defs/pipeline_component"
|
|
306
|
+
},
|
|
307
|
+
"transformer_2": {
|
|
308
|
+
"$ref": "#/$defs/pipeline_component"
|
|
309
|
+
},
|
|
310
|
+
"vae": {
|
|
311
|
+
"$ref": "#/$defs/pipeline_component"
|
|
312
|
+
},
|
|
313
|
+
"unet": {
|
|
314
|
+
"$ref": "#/$defs/pipeline_component"
|
|
315
|
+
},
|
|
316
|
+
"text_encoder": {
|
|
317
|
+
"$ref": "#/$defs/pipeline_component"
|
|
318
|
+
},
|
|
319
|
+
"text_encoder_2": {
|
|
320
|
+
"$ref": "#/$defs/pipeline_component"
|
|
321
|
+
},
|
|
322
|
+
"text_encoder_3": {
|
|
323
|
+
"$ref": "#/$defs/pipeline_component"
|
|
324
|
+
},
|
|
325
|
+
"tokenizer": {
|
|
326
|
+
"$ref": "#/$defs/pipeline_component"
|
|
327
|
+
},
|
|
328
|
+
"tokenizer_2": {
|
|
329
|
+
"$ref": "#/$defs/pipeline_component"
|
|
330
|
+
},
|
|
331
|
+
"tokenizer_3": {
|
|
332
|
+
"$ref": "#/$defs/pipeline_component"
|
|
333
|
+
},
|
|
334
|
+
"image_encoder": {
|
|
335
|
+
"$ref": "#/$defs/pipeline_component"
|
|
336
|
+
},
|
|
337
|
+
"feature_extractor": {
|
|
338
|
+
"$ref": "#/$defs/pipeline_component"
|
|
339
|
+
},
|
|
340
|
+
"prompt_enhancer_head": {
|
|
341
|
+
"$ref": "#/$defs/pipeline_component"
|
|
342
|
+
},
|
|
343
|
+
"controlnet": {
|
|
344
|
+
"$ref": "#/$defs/controlnet"
|
|
345
|
+
},
|
|
346
|
+
"loras": {
|
|
347
|
+
"type": "array",
|
|
348
|
+
"items": {
|
|
349
|
+
"$ref": "#/$defs/lora"
|
|
350
|
+
}
|
|
351
|
+
},
|
|
352
|
+
"ip_adapter": {
|
|
353
|
+
"$ref": "#/$defs/ip_adapter"
|
|
354
|
+
},
|
|
355
|
+
"from_pretrained_arguments": {
|
|
356
|
+
"$ref": "#/$defs/from_pretrained_arguments"
|
|
357
|
+
},
|
|
358
|
+
"remote_text_encoder": {
|
|
359
|
+
"description": "Encode prompts with a remote text-encoder service instead of loading the text encoder locally. The pipeline's text_encoder is set to None, the prompt is POSTed to the service (authenticated with the HuggingFace token), and the returned embeddings are passed to the pipeline as prompt_embeds. Mutually exclusive with prompt_weighting.",
|
|
360
|
+
"type": "object",
|
|
361
|
+
"properties": {
|
|
362
|
+
"url": {
|
|
363
|
+
"description": "URL of the remote text encoder endpoint, e.g. a HuggingFace inference endpoint.",
|
|
364
|
+
"type": "string"
|
|
365
|
+
}
|
|
366
|
+
},
|
|
367
|
+
"required": [
|
|
368
|
+
"url"
|
|
369
|
+
]
|
|
370
|
+
},
|
|
371
|
+
"seed": {
|
|
372
|
+
"description": "The seed for this pipeline",
|
|
373
|
+
"type": "integer",
|
|
374
|
+
"format": "int64"
|
|
375
|
+
},
|
|
376
|
+
"chain": {
|
|
377
|
+
"$ref": "#/$defs/chain"
|
|
378
|
+
},
|
|
379
|
+
"arguments": {
|
|
380
|
+
"$ref": "#/$defs/arguments"
|
|
381
|
+
}
|
|
382
|
+
},
|
|
383
|
+
"required": [
|
|
384
|
+
"configuration",
|
|
385
|
+
"from_pretrained_arguments",
|
|
386
|
+
"arguments"
|
|
387
|
+
]
|
|
388
|
+
},
|
|
389
|
+
"pipeline_configuration": {
|
|
390
|
+
"type": "object",
|
|
391
|
+
"properties": {
|
|
392
|
+
"offload": {
|
|
393
|
+
"type": "string",
|
|
394
|
+
"enum": [
|
|
395
|
+
"model",
|
|
396
|
+
"sequential"
|
|
397
|
+
]
|
|
398
|
+
},
|
|
399
|
+
"exclude_from_cpu_offload": {
|
|
400
|
+
"description": "Component names (e.g. 'vae', 'text_encoder') to keep resident on the accelerator instead of offloading. Only applies when offload is 'sequential'.",
|
|
401
|
+
"type": "array",
|
|
402
|
+
"items": {
|
|
403
|
+
"type": "string"
|
|
404
|
+
}
|
|
405
|
+
},
|
|
406
|
+
"group_offload": {
|
|
407
|
+
"$ref": "#/$defs/group_offload"
|
|
408
|
+
},
|
|
409
|
+
"device": {
|
|
410
|
+
"description": "The device to run this pipeline on, e.g. 'cuda', 'cuda:1', 'mps' or 'cpu'. Defaults to the device dw is running on and becomes the default for this pipeline's components.",
|
|
411
|
+
"type": "string"
|
|
412
|
+
},
|
|
413
|
+
"component_type": {
|
|
414
|
+
"description": "The python type of pipeline to use in the format 'module.typename' module defaults to diffusers",
|
|
415
|
+
"type": "string"
|
|
416
|
+
},
|
|
417
|
+
"no_generator": {
|
|
418
|
+
"description": "Whether to use a generator for the pipeline. Some pipelines do not support generators.",
|
|
419
|
+
"type": "boolean"
|
|
420
|
+
},
|
|
421
|
+
"enable_attention_slicing": {
|
|
422
|
+
"description": "Whether to enable attention slicing for the pipeline to reduce memory usage.",
|
|
423
|
+
"type": "boolean"
|
|
424
|
+
},
|
|
425
|
+
"disable_attention_slicing": {
|
|
426
|
+
"description": "Whether to disable automatic attention slicing on MPS devices. Attention slicing is enabled by default on MPS to reduce memory usage.",
|
|
427
|
+
"type": "boolean"
|
|
428
|
+
},
|
|
429
|
+
"attention_backend": {
|
|
430
|
+
"description": "The attention backend to use for the pipeline.",
|
|
431
|
+
"type": "string"
|
|
432
|
+
},
|
|
433
|
+
"prompt_weighting": {
|
|
434
|
+
"description": "Enable A1111-style prompt weighting syntax: (word:1.5) for emphasis, [word] for de-emphasis, ((word)) for nested weighting. Also supports prompts longer than 77 tokens. Currently supports Flux pipelines. Mutually exclusive with remote_text_encoder.",
|
|
435
|
+
"type": "boolean"
|
|
436
|
+
},
|
|
437
|
+
"pre_load_modules": {
|
|
438
|
+
"description": "List of Python modules to import before loading the pipeline. Used for modules that register with diffusers on import (e.g., sdnq).",
|
|
439
|
+
"type": "array",
|
|
440
|
+
"items": {
|
|
441
|
+
"type": "string"
|
|
442
|
+
}
|
|
443
|
+
},
|
|
444
|
+
"load_components": {
|
|
445
|
+
"description": "Arguments passed to a modular pipeline's load_components() - modular pipelines load their component weights separately from their config. Use 'dtype' to set the component dtype and 'names' to load a subset of components.",
|
|
446
|
+
"type": "object",
|
|
447
|
+
"properties": {
|
|
448
|
+
"names": {
|
|
449
|
+
"description": "Names of the components to load. All components are loaded when omitted.",
|
|
450
|
+
"type": "array",
|
|
451
|
+
"items": {
|
|
452
|
+
"type": "string"
|
|
453
|
+
}
|
|
454
|
+
},
|
|
455
|
+
"dtype": {
|
|
456
|
+
"description": "The torch dtype to load components in, e.g. 'torch.bfloat16'",
|
|
457
|
+
"type": "string"
|
|
458
|
+
},
|
|
459
|
+
"quantization_config": {
|
|
460
|
+
"description": "Quantization configuration per component, keyed by component name (e.g. 'transformer', 'text_encoder'). A component not named here loads unquantized.",
|
|
461
|
+
"type": "object",
|
|
462
|
+
"additionalProperties": {
|
|
463
|
+
"$ref": "#/$defs/quantization_config"
|
|
464
|
+
}
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
},
|
|
468
|
+
"configs": {
|
|
469
|
+
"description": "Values a modular pipeline's blocks declare and read while they run - not components, and not call arguments. Keys are whatever the pipeline itself declares, so what may be set here depends on the model: MiniMax-H3 declares 'canvas_short_edge', 'canvas_max_pixels' and 'reference_image_short_edge'. A name the pipeline does not declare is an error rather than a setting that quietly did nothing.",
|
|
470
|
+
"type": "object",
|
|
471
|
+
"additionalProperties": {
|
|
472
|
+
"type": [
|
|
473
|
+
"string",
|
|
474
|
+
"integer",
|
|
475
|
+
"number",
|
|
476
|
+
"boolean",
|
|
477
|
+
"array",
|
|
478
|
+
"null"
|
|
479
|
+
]
|
|
480
|
+
}
|
|
481
|
+
},
|
|
482
|
+
"components": {
|
|
483
|
+
"description": "Configuration applied to components once the pipeline has loaded them. This is where a modular pipeline's components are placed, since it pulls their weights itself. Keys name a component, optionally as a dotted path into one, e.g. 'text_encoder.model'.",
|
|
484
|
+
"type": "object",
|
|
485
|
+
"additionalProperties": {
|
|
486
|
+
"type": "object",
|
|
487
|
+
"properties": {
|
|
488
|
+
"group_offload": {
|
|
489
|
+
"$ref": "#/$defs/group_offload"
|
|
490
|
+
},
|
|
491
|
+
"device": {
|
|
492
|
+
"description": "The device to move this component to, e.g. 'cuda'. Only for components small enough to stay resident - the offloaded ones are placed by their own hooks.",
|
|
493
|
+
"type": "string"
|
|
494
|
+
},
|
|
495
|
+
"residency": {
|
|
496
|
+
"description": "Whether the component stays on its device for the whole run ('resident', the default) or rests in system memory and is moved to the device only while one of its own calls runs ('on_demand'). Use on_demand for a component that is large but called a handful of times, like a VAE that only encodes references and decodes the result - it frees the device for the rest of the run and, unlike group_offload, moves the model whole, so a tiled decode costs one pair of transfers rather than one per tile. Not for a component called every step, and cannot be combined with group_offload.",
|
|
497
|
+
"type": "string",
|
|
498
|
+
"enum": [
|
|
499
|
+
"resident",
|
|
500
|
+
"on_demand"
|
|
501
|
+
]
|
|
502
|
+
},
|
|
503
|
+
"enable_tiling": {
|
|
504
|
+
"description": "Enable tiled decoding on this component, for a decoder that is not the one named 'vae' (which the pipeline-level 'vae' block covers) - LTX-2.5's 'diffusion_decoder', say, which otherwise decodes the whole video volume in one allocation. 'true' uses the model's own default tile size; an object passes tile and stride sizes through to enable_tiling(), which is what a card smaller than those defaults needs.",
|
|
505
|
+
"type": [
|
|
506
|
+
"boolean",
|
|
507
|
+
"object"
|
|
508
|
+
]
|
|
509
|
+
},
|
|
510
|
+
"attention_backend": {
|
|
511
|
+
"description": "Pin this component's attention backend persistently via set_attention_backend, e.g. 'flash', 'sage', '_flash_3_hub'. Use this instead of the pipeline-level attention_backend when the component is compiled - the per-call context manager forces recompiles.",
|
|
512
|
+
"type": "string"
|
|
513
|
+
},
|
|
514
|
+
"truncate_layers": {
|
|
515
|
+
"description": "Drop the tail of a ModuleList inside this component that the run never reads, before any offload hooks are installed. Keys are dotted paths relative to the component, values the number of entries to keep. For an encoder used for its hidden states: MiniMax-H3 conditions on hidden_states[50] of its 64-layer Qwen3-VL, so { 'language_model.layers': 51 } drops the 13 layers whose output nothing consumes while leaving hidden_states[50] bit-identical (keeping only 50 would hand back the final-norm output instead, a different tensor).",
|
|
516
|
+
"type": "object",
|
|
517
|
+
"additionalProperties": {
|
|
518
|
+
"type": "integer",
|
|
519
|
+
"minimum": 1
|
|
520
|
+
}
|
|
521
|
+
},
|
|
522
|
+
"remove_modules": {
|
|
523
|
+
"description": "Dotted paths to modules inside this component that the run never calls, each replaced with an Identity before any offload hooks are installed - a language-model head on a model used as an encoder, say. The attribute survives for anything that looks it up; the weights are freed rather than held (and, offloaded, pinned) for a call that never comes.",
|
|
524
|
+
"type": "array",
|
|
525
|
+
"items": {
|
|
526
|
+
"type": "string"
|
|
527
|
+
}
|
|
528
|
+
},
|
|
529
|
+
"compile": {
|
|
530
|
+
"$ref": "#/$defs/compile_config"
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
},
|
|
535
|
+
"preserve_device_placement": {
|
|
536
|
+
"description": "Whether to leave the component where it loaded rather than moving it to the device. Needed when the component is loaded already placed - by a 'device_map', or by a quantization that pins its tensors to one device. A 'components' block that group offloads already keeps the pipeline off the device on its own. Renamed from 'do_not_send_to_device', which is no longer recognized.",
|
|
537
|
+
"type": "boolean"
|
|
538
|
+
},
|
|
539
|
+
"components_manager": {
|
|
540
|
+
"description": "Attach a ComponentsManager to a modular pipeline. The manager tracks the pipeline's components and, with auto CPU offload enabled, keeps only the running ones on the device. It replaces 'offload', which modular pipelines do not support.",
|
|
541
|
+
"type": "object",
|
|
542
|
+
"properties": {
|
|
543
|
+
"enable_auto_cpu_offload": {
|
|
544
|
+
"description": "Move components between the device and system memory as the pipeline runs. Requires a device that reports free memory (CUDA). When enabled the manager owns device placement, so the pipeline is not moved to the device directly.",
|
|
545
|
+
"type": "boolean",
|
|
546
|
+
"default": false
|
|
547
|
+
},
|
|
548
|
+
"memory_reserve_margin": {
|
|
549
|
+
"description": "Device memory to keep free when deciding what to offload, e.g. '3GB'",
|
|
550
|
+
"type": "string",
|
|
551
|
+
"default": "3GB"
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
},
|
|
555
|
+
"sdnq_optimize": {
|
|
556
|
+
"description": "List of pipeline component names to apply SDNQ quantized matmul optimization to (e.g., ['transformer', 'text_encoder']). Requires sdnq package and CUDA/XPU hardware.",
|
|
557
|
+
"type": "array",
|
|
558
|
+
"items": {
|
|
559
|
+
"type": "string"
|
|
560
|
+
}
|
|
561
|
+
},
|
|
562
|
+
"cache": {
|
|
563
|
+
"description": "Enable diffusers built-in cache acceleration on the transformer. Mutually exclusive with teacache. Hooks auto-reset between inference runs.",
|
|
564
|
+
"type": "object",
|
|
565
|
+
"properties": {
|
|
566
|
+
"type": {
|
|
567
|
+
"description": "Cache algorithm: first_block (simplest, broadest support), faster (video-oriented, experimental), mag (magnitude-based, needs num_inference_steps), taylorseer (Taylor series approximation), text_kv (text key-value cache).",
|
|
568
|
+
"type": "string",
|
|
569
|
+
"enum": [
|
|
570
|
+
"first_block",
|
|
571
|
+
"faster",
|
|
572
|
+
"mag",
|
|
573
|
+
"taylorseer",
|
|
574
|
+
"text_kv"
|
|
575
|
+
]
|
|
576
|
+
},
|
|
577
|
+
"threshold": {
|
|
578
|
+
"description": "Cache threshold for first_block and mag types. Higher = more speedup, more quality loss. first_block default: 0.05, mag default: 0.06. Accepts a 'variable:' reference.",
|
|
579
|
+
"type": [
|
|
580
|
+
"number",
|
|
581
|
+
"string"
|
|
582
|
+
]
|
|
583
|
+
},
|
|
584
|
+
"num_inference_steps": {
|
|
585
|
+
"description": "Required for mag cache type. Must match the pipeline's num_inference_steps.",
|
|
586
|
+
"type": "integer"
|
|
587
|
+
},
|
|
588
|
+
"max_skip_steps": {
|
|
589
|
+
"description": "Max consecutive skippable steps for mag cache. Default: 3.",
|
|
590
|
+
"type": "integer"
|
|
591
|
+
},
|
|
592
|
+
"retention_ratio": {
|
|
593
|
+
"description": "Fraction of initial steps where skipping is disabled for mag cache. Default: 0.2.",
|
|
594
|
+
"type": "number"
|
|
595
|
+
},
|
|
596
|
+
"mag_ratios": {
|
|
597
|
+
"description": "Per-step magnitude ratios for mag cache. Required unless calibrate is true, and checkpoint-dependent. Either a preset name shipped by diffusers ('flux') or an explicit array of per-step ratios. Interpolated automatically when its length differs from num_inference_steps.",
|
|
598
|
+
"oneOf": [
|
|
599
|
+
{
|
|
600
|
+
"type": "string"
|
|
601
|
+
},
|
|
602
|
+
{
|
|
603
|
+
"type": "array",
|
|
604
|
+
"items": {
|
|
605
|
+
"type": "number"
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
]
|
|
609
|
+
},
|
|
610
|
+
"calibrate": {
|
|
611
|
+
"description": "Run mag cache in calibration mode: skip nothing, and log the magnitude ratios for this model at the end of the run so they can be pasted into mag_ratios. Default: false.",
|
|
612
|
+
"type": "boolean"
|
|
613
|
+
},
|
|
614
|
+
"cache_interval": {
|
|
615
|
+
"description": "Full computation every N steps for taylorseer cache. Default: 5.",
|
|
616
|
+
"type": "integer"
|
|
617
|
+
},
|
|
618
|
+
"max_order": {
|
|
619
|
+
"description": "Taylor series order for taylorseer cache. Higher = better approximation, more memory. Default: 1.",
|
|
620
|
+
"type": "integer"
|
|
621
|
+
}
|
|
622
|
+
},
|
|
623
|
+
"required": [
|
|
624
|
+
"type"
|
|
625
|
+
]
|
|
626
|
+
},
|
|
627
|
+
"teacache": {
|
|
628
|
+
"description": "Enable TeaCache inference acceleration. Auto-detects transformer type. Caches intermediate computations and skips redundant steps. Requires num_inference_steps in pipeline arguments. Currently supports FluxTransformer2DModel. Mutually exclusive with cache.",
|
|
629
|
+
"type": "object",
|
|
630
|
+
"properties": {
|
|
631
|
+
"rel_l1_thresh": {
|
|
632
|
+
"description": "Cache threshold. Higher = more speedup, more quality loss. Model-specific defaults apply if omitted.",
|
|
633
|
+
"type": "number"
|
|
634
|
+
},
|
|
635
|
+
"coefficients": {
|
|
636
|
+
"description": "Polynomial coefficients for rescaling relative L1 distance. 5 floats for np.poly1d. If omitted, uses model-specific defaults.",
|
|
637
|
+
"type": "array",
|
|
638
|
+
"items": {
|
|
639
|
+
"type": "number"
|
|
640
|
+
},
|
|
641
|
+
"minItems": 5,
|
|
642
|
+
"maxItems": 5
|
|
643
|
+
},
|
|
644
|
+
"variant": {
|
|
645
|
+
"description": "Explicit model variant from teacache_models.json. Required for models with multiple variants (e.g., 'cogvideox_2b', 'wan2.1_t2v_1.3b'). If omitted, uses the class default.",
|
|
646
|
+
"type": "string"
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
},
|
|
650
|
+
"vae": {
|
|
651
|
+
"type": "object",
|
|
652
|
+
"properties": {
|
|
653
|
+
"enable_slicing": {
|
|
654
|
+
"type": "boolean"
|
|
655
|
+
},
|
|
656
|
+
"enable_tiling": {
|
|
657
|
+
"type": "boolean"
|
|
658
|
+
},
|
|
659
|
+
"channels_last": {
|
|
660
|
+
"type": "boolean"
|
|
661
|
+
},
|
|
662
|
+
"torch_dtype": {
|
|
663
|
+
"type": "string"
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
},
|
|
667
|
+
"unet": {
|
|
668
|
+
"type": "object",
|
|
669
|
+
"properties": {
|
|
670
|
+
"enable_forward_chunking": {
|
|
671
|
+
"type": "boolean"
|
|
672
|
+
},
|
|
673
|
+
"channels_last": {
|
|
674
|
+
"type": "boolean"
|
|
675
|
+
},
|
|
676
|
+
"torch_dtype": {
|
|
677
|
+
"type": "string"
|
|
678
|
+
},
|
|
679
|
+
"attn_processor_type": {
|
|
680
|
+
"type": "string"
|
|
681
|
+
}
|
|
682
|
+
}
|
|
683
|
+
},
|
|
684
|
+
"transformer": {
|
|
685
|
+
"type": "object",
|
|
686
|
+
"properties": {
|
|
687
|
+
"attn_processor_type": {
|
|
688
|
+
"description": "The type name of an attention processor",
|
|
689
|
+
"type": "string"
|
|
690
|
+
}
|
|
691
|
+
}
|
|
692
|
+
},
|
|
693
|
+
"text_encoder": {
|
|
694
|
+
"type": "object",
|
|
695
|
+
"properties": {
|
|
696
|
+
"torch_dtype": {
|
|
697
|
+
"type": "string"
|
|
698
|
+
}
|
|
699
|
+
}
|
|
700
|
+
},
|
|
701
|
+
"shared_components": {
|
|
702
|
+
"$ref": "#/$defs/shared_components"
|
|
703
|
+
},
|
|
704
|
+
"reused_components": {
|
|
705
|
+
"$ref": "#/$defs/reused_components"
|
|
706
|
+
},
|
|
707
|
+
"enable_layerwise_casting": {
|
|
708
|
+
"$ref": "#/$defs/enable_layerwise_casting"
|
|
709
|
+
},
|
|
710
|
+
"inversion": {
|
|
711
|
+
"description": "Run the pipeline's invert() rather than the pipeline itself, returning the inverted latents. Only for pipelines that have one.",
|
|
712
|
+
"type": "boolean"
|
|
713
|
+
},
|
|
714
|
+
"generate": {
|
|
715
|
+
"description": "Run the pipeline's generate() rather than the pipeline itself, returning the generated ids. Only for pipelines that have one, such as a conditioner used on its own.",
|
|
716
|
+
"type": "boolean"
|
|
717
|
+
}
|
|
718
|
+
},
|
|
719
|
+
"additionalProperties": false,
|
|
720
|
+
"required": [
|
|
721
|
+
"component_type"
|
|
722
|
+
]
|
|
723
|
+
},
|
|
724
|
+
"shared_components": {
|
|
725
|
+
"description": "The names of components this step loads and later steps may reuse. A modular pipeline is given them after it is built, so the components it shares are the ones it actually loaded.",
|
|
726
|
+
"type": "array",
|
|
727
|
+
"items": {
|
|
728
|
+
"type": "string"
|
|
729
|
+
}
|
|
730
|
+
},
|
|
731
|
+
"reused_components": {
|
|
732
|
+
"description": "The names of components an earlier step shared, reused here instead of being loaded again. A reused component keeps the device placement the step that shared it gave it.",
|
|
733
|
+
"type": "array",
|
|
734
|
+
"items": {
|
|
735
|
+
"type": "string"
|
|
736
|
+
}
|
|
737
|
+
},
|
|
738
|
+
"scheduler": {
|
|
739
|
+
"type": "object",
|
|
740
|
+
"properties": {
|
|
741
|
+
"configuration": {
|
|
742
|
+
"type": "object",
|
|
743
|
+
"properties": {
|
|
744
|
+
"scheduler_type": {
|
|
745
|
+
"type": "string"
|
|
746
|
+
}
|
|
747
|
+
},
|
|
748
|
+
"required": [
|
|
749
|
+
"scheduler_type"
|
|
750
|
+
]
|
|
751
|
+
},
|
|
752
|
+
"from_config_args": {
|
|
753
|
+
"$ref": "#/$defs/arguments"
|
|
754
|
+
},
|
|
755
|
+
"shift": {
|
|
756
|
+
"description": "Exponential sigma shift applied to this scheduler's schedule, for schedulers that take one (MiniMax H3: 12.0 for video, 3.0 for audio in the released checkpoint). Lower it when running few steps - a high shift packs the whole grid near sigma 1 and leaves one enormous final step, which a short schedule cannot absorb. Accepts a 'variable:' reference.",
|
|
757
|
+
"type": [
|
|
758
|
+
"number",
|
|
759
|
+
"string"
|
|
760
|
+
],
|
|
761
|
+
"exclusiveMinimum": 0
|
|
762
|
+
}
|
|
763
|
+
},
|
|
764
|
+
"anyOf": [
|
|
765
|
+
{
|
|
766
|
+
"required": [
|
|
767
|
+
"configuration"
|
|
768
|
+
]
|
|
769
|
+
},
|
|
770
|
+
{
|
|
771
|
+
"required": [
|
|
772
|
+
"shift"
|
|
773
|
+
]
|
|
774
|
+
}
|
|
775
|
+
]
|
|
776
|
+
},
|
|
777
|
+
"pipeline_component": {
|
|
778
|
+
"type": "object",
|
|
779
|
+
"properties": {
|
|
780
|
+
"configuration": {
|
|
781
|
+
"type": "object",
|
|
782
|
+
"properties": {
|
|
783
|
+
"device": {
|
|
784
|
+
"description": "The device to load this component onto, e.g. 'cuda', 'cuda:1', 'mps' or 'cpu'. Defaults to the device its pipeline runs on.",
|
|
785
|
+
"type": "string"
|
|
786
|
+
},
|
|
787
|
+
"component_type": {
|
|
788
|
+
"description": "The python type of pipeline component to use in the format 'module.typename' module defaults to diffusers",
|
|
789
|
+
"type": "string"
|
|
790
|
+
}
|
|
791
|
+
},
|
|
792
|
+
"required": [
|
|
793
|
+
"component_type"
|
|
794
|
+
]
|
|
795
|
+
},
|
|
796
|
+
"group_offload": {
|
|
797
|
+
"$ref": "#/$defs/group_offload"
|
|
798
|
+
},
|
|
799
|
+
"enable_layerwise_casting": {
|
|
800
|
+
"$ref": "#/$defs/enable_layerwise_casting"
|
|
801
|
+
},
|
|
802
|
+
"quantization_config": {
|
|
803
|
+
"$ref": "#/$defs/quantization_config"
|
|
804
|
+
},
|
|
805
|
+
"from_pretrained_arguments": {
|
|
806
|
+
"$ref": "#/$defs/from_pretrained_arguments"
|
|
807
|
+
}
|
|
808
|
+
},
|
|
809
|
+
"required": [
|
|
810
|
+
"configuration",
|
|
811
|
+
"from_pretrained_arguments"
|
|
812
|
+
]
|
|
813
|
+
},
|
|
814
|
+
"quantization_config": {
|
|
815
|
+
"type": "object",
|
|
816
|
+
"properties": {
|
|
817
|
+
"configuration": {
|
|
818
|
+
"type": "object",
|
|
819
|
+
"properties": {
|
|
820
|
+
"config_type": {
|
|
821
|
+
"type": "string"
|
|
822
|
+
}
|
|
823
|
+
},
|
|
824
|
+
"required": [
|
|
825
|
+
"config_type"
|
|
826
|
+
]
|
|
827
|
+
},
|
|
828
|
+
"arguments": {
|
|
829
|
+
"$ref": "#/$defs/arguments"
|
|
830
|
+
}
|
|
831
|
+
},
|
|
832
|
+
"required": [
|
|
833
|
+
"configuration",
|
|
834
|
+
"arguments"
|
|
835
|
+
]
|
|
836
|
+
},
|
|
837
|
+
"controlnet": {
|
|
838
|
+
"type": "object",
|
|
839
|
+
"properties": {
|
|
840
|
+
"configuration": {
|
|
841
|
+
"$ref": "#/$defs/pipeline_configuration"
|
|
842
|
+
},
|
|
843
|
+
"from_pretrained_arguments": {
|
|
844
|
+
"$ref": "#/$defs/from_pretrained_arguments"
|
|
845
|
+
}
|
|
846
|
+
},
|
|
847
|
+
"required": [
|
|
848
|
+
"configuration",
|
|
849
|
+
"from_pretrained_arguments"
|
|
850
|
+
]
|
|
851
|
+
},
|
|
852
|
+
"ip_adapter": {
|
|
853
|
+
"type": "object",
|
|
854
|
+
"properties": {
|
|
855
|
+
"model_name": {
|
|
856
|
+
"description": "The huggingface hub name of the ip adadapter model to use.",
|
|
857
|
+
"type": "string"
|
|
858
|
+
},
|
|
859
|
+
"weight_name": {
|
|
860
|
+
"description": "The file name of the ip adapter weights.",
|
|
861
|
+
"type": "string"
|
|
862
|
+
},
|
|
863
|
+
"subfolder": {
|
|
864
|
+
"description": "The subfolder location of a model file within a larger model repository",
|
|
865
|
+
"type": [
|
|
866
|
+
"string",
|
|
867
|
+
"null"
|
|
868
|
+
]
|
|
869
|
+
},
|
|
870
|
+
"scale": {
|
|
871
|
+
"description": "The scale factor of the ip adapter.",
|
|
872
|
+
"type": "number",
|
|
873
|
+
"format": "float"
|
|
874
|
+
}
|
|
875
|
+
},
|
|
876
|
+
"required": [
|
|
877
|
+
"model_name"
|
|
878
|
+
]
|
|
879
|
+
},
|
|
880
|
+
"lora": {
|
|
881
|
+
"type": "object",
|
|
882
|
+
"properties": {
|
|
883
|
+
"model_name": {
|
|
884
|
+
"description": "The huggingface hub name of the lora.",
|
|
885
|
+
"type": "string"
|
|
886
|
+
},
|
|
887
|
+
"weight_name": {
|
|
888
|
+
"description": "The file name of the lora weights.",
|
|
889
|
+
"type": "string"
|
|
890
|
+
},
|
|
891
|
+
"subfolder": {
|
|
892
|
+
"description": "The subfolder location of a model file within a larger model repository",
|
|
893
|
+
"type": "string"
|
|
894
|
+
},
|
|
895
|
+
"scale": {
|
|
896
|
+
"description": "The scale factor of the lora when fusing. Accepts a 'variable:' reference.",
|
|
897
|
+
"type": [
|
|
898
|
+
"number",
|
|
899
|
+
"string"
|
|
900
|
+
],
|
|
901
|
+
"format": "float"
|
|
902
|
+
},
|
|
903
|
+
"adapter_name": {
|
|
904
|
+
"description": "The user defined name of the adapter to pass to set_adapters function",
|
|
905
|
+
"type": "string"
|
|
906
|
+
}
|
|
907
|
+
},
|
|
908
|
+
"required": [
|
|
909
|
+
"model_name"
|
|
910
|
+
]
|
|
911
|
+
},
|
|
912
|
+
"from_pretrained_arguments": {
|
|
913
|
+
"type": "object",
|
|
914
|
+
"description": "Arguments to pass to the from_pretrained function when creating the component. Names at most one source: 'model_name', 'from_single_file' or 'task'. A pipeline assembled entirely from components an earlier step shared and components declared beside it - LTX-2's latent upsampler, say - names none of them and is constructed directly from those components.",
|
|
915
|
+
"properties": {
|
|
916
|
+
"model_name": {
|
|
917
|
+
"description": "The huggingface hub name of the pretrained model to load",
|
|
918
|
+
"type": "string"
|
|
919
|
+
},
|
|
920
|
+
"from_single_file": {
|
|
921
|
+
"description": "Location of a checkpoint model from a single file",
|
|
922
|
+
"type": "string"
|
|
923
|
+
},
|
|
924
|
+
"task": {
|
|
925
|
+
"description": "The name of a task that transformers.Pipeline understands",
|
|
926
|
+
"type": "string"
|
|
927
|
+
}
|
|
928
|
+
},
|
|
929
|
+
"additionalProperties": {
|
|
930
|
+
"type": [
|
|
931
|
+
"string",
|
|
932
|
+
"number",
|
|
933
|
+
"object",
|
|
934
|
+
"array",
|
|
935
|
+
"boolean",
|
|
936
|
+
"null"
|
|
937
|
+
]
|
|
938
|
+
},
|
|
939
|
+
"oneOf": [
|
|
940
|
+
{
|
|
941
|
+
"required": [
|
|
942
|
+
"model_name"
|
|
943
|
+
]
|
|
944
|
+
},
|
|
945
|
+
{
|
|
946
|
+
"required": [
|
|
947
|
+
"from_single_file"
|
|
948
|
+
]
|
|
949
|
+
},
|
|
950
|
+
{
|
|
951
|
+
"required": [
|
|
952
|
+
"task"
|
|
953
|
+
]
|
|
954
|
+
},
|
|
955
|
+
{
|
|
956
|
+
"allOf": [
|
|
957
|
+
{
|
|
958
|
+
"not": {
|
|
959
|
+
"required": [
|
|
960
|
+
"model_name"
|
|
961
|
+
]
|
|
962
|
+
}
|
|
963
|
+
},
|
|
964
|
+
{
|
|
965
|
+
"not": {
|
|
966
|
+
"required": [
|
|
967
|
+
"from_single_file"
|
|
968
|
+
]
|
|
969
|
+
}
|
|
970
|
+
},
|
|
971
|
+
{
|
|
972
|
+
"not": {
|
|
973
|
+
"required": [
|
|
974
|
+
"task"
|
|
975
|
+
]
|
|
976
|
+
}
|
|
977
|
+
}
|
|
978
|
+
]
|
|
979
|
+
}
|
|
980
|
+
]
|
|
981
|
+
},
|
|
982
|
+
"task": {
|
|
983
|
+
"type": "object",
|
|
984
|
+
"properties": {
|
|
985
|
+
"command": {
|
|
986
|
+
"type": "string"
|
|
987
|
+
},
|
|
988
|
+
"arguments": {
|
|
989
|
+
"$ref": "#/$defs/arguments"
|
|
990
|
+
},
|
|
991
|
+
"inputs": {
|
|
992
|
+
"type": "array",
|
|
993
|
+
"items": {
|
|
994
|
+
"type": [
|
|
995
|
+
"string",
|
|
996
|
+
"integer",
|
|
997
|
+
"number",
|
|
998
|
+
"object",
|
|
999
|
+
"boolean"
|
|
1000
|
+
]
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
1003
|
+
},
|
|
1004
|
+
"oneOf": [
|
|
1005
|
+
{
|
|
1006
|
+
"required": [
|
|
1007
|
+
"command",
|
|
1008
|
+
"arguments"
|
|
1009
|
+
]
|
|
1010
|
+
},
|
|
1011
|
+
{
|
|
1012
|
+
"required": [
|
|
1013
|
+
"command",
|
|
1014
|
+
"inputs"
|
|
1015
|
+
]
|
|
1016
|
+
}
|
|
1017
|
+
]
|
|
1018
|
+
},
|
|
1019
|
+
"workflow_reference": {
|
|
1020
|
+
"type": "object",
|
|
1021
|
+
"properties": {
|
|
1022
|
+
"path": {
|
|
1023
|
+
"description": "The path to the workflow file. Use 'builtin:' for built-in workflows.",
|
|
1024
|
+
"type": "string"
|
|
1025
|
+
},
|
|
1026
|
+
"arguments": {
|
|
1027
|
+
"$ref": "#/$defs/arguments"
|
|
1028
|
+
}
|
|
1029
|
+
}
|
|
1030
|
+
},
|
|
1031
|
+
"result": {
|
|
1032
|
+
"type": "object",
|
|
1033
|
+
"properties": {
|
|
1034
|
+
"content_type": {
|
|
1035
|
+
"description": "The content type of the result when serialized to disk. Audio can be written as 'audio/wav', 'audio/flac', 'audio/mpeg' (mp3), 'audio/ogg' or 'audio/opus'. Opus is written into an ogg container and only encodes sample rates of 8000, 12000, 16000, 24000 or 48000.",
|
|
1036
|
+
"type": "string"
|
|
1037
|
+
},
|
|
1038
|
+
"save": {
|
|
1039
|
+
"description": "Whether to save the result",
|
|
1040
|
+
"type": "boolean",
|
|
1041
|
+
"default": true
|
|
1042
|
+
},
|
|
1043
|
+
"file_base_name": {
|
|
1044
|
+
"description": "The base name for saving result and metadata files. The file extension is determined by context and content_type.",
|
|
1045
|
+
"type": "string"
|
|
1046
|
+
},
|
|
1047
|
+
"fps": {
|
|
1048
|
+
"description": "Frames per second - only used when output is video",
|
|
1049
|
+
"type": "integer",
|
|
1050
|
+
"default": 8
|
|
1051
|
+
},
|
|
1052
|
+
"sample_rate": {
|
|
1053
|
+
"description": "Audio sample rate - only used when output is audio",
|
|
1054
|
+
"type": "integer",
|
|
1055
|
+
"default": 44100
|
|
1056
|
+
},
|
|
1057
|
+
"audio_sample_rate": {
|
|
1058
|
+
"description": "Sample rate of the audio generated with a video, muxed into 'video/mp4' output. Defaults to the rate reported by the pipeline's vocoder - only set this to override it.",
|
|
1059
|
+
"type": "integer"
|
|
1060
|
+
},
|
|
1061
|
+
"subtype": {
|
|
1062
|
+
"description": "Audio encoding subtype, e.g. 'PCM_24' for wav and flac or 'VORBIS' for ogg. Defaults to the container's own default ('PCM_16' for wav and flac). Only used when output is audio.",
|
|
1063
|
+
"type": "string"
|
|
1064
|
+
},
|
|
1065
|
+
"compression_level": {
|
|
1066
|
+
"description": "Encoding compression level from 0.0 to 1.0 for compressed audio formats (flac, mp3, ogg). Higher means smaller files. Only used when output is audio.",
|
|
1067
|
+
"type": "number",
|
|
1068
|
+
"minimum": 0,
|
|
1069
|
+
"maximum": 1
|
|
1070
|
+
},
|
|
1071
|
+
"bitrate_mode": {
|
|
1072
|
+
"description": "Bitrate mode for compressed audio formats: 'CONSTANT', 'AVERAGE' or 'VARIABLE'. Only used when output is audio.",
|
|
1073
|
+
"type": "string",
|
|
1074
|
+
"enum": [
|
|
1075
|
+
"CONSTANT",
|
|
1076
|
+
"AVERAGE",
|
|
1077
|
+
"VARIABLE"
|
|
1078
|
+
]
|
|
1079
|
+
},
|
|
1080
|
+
"embed_metadata": {
|
|
1081
|
+
"description": "Whether to embed generation parameters as metadata in saved images (PNG info chunks or EXIF). Only applies to image content types.",
|
|
1082
|
+
"type": "boolean",
|
|
1083
|
+
"default": false
|
|
1084
|
+
}
|
|
1085
|
+
},
|
|
1086
|
+
"additionalProperties": {
|
|
1087
|
+
"type": [
|
|
1088
|
+
"string",
|
|
1089
|
+
"number",
|
|
1090
|
+
"object",
|
|
1091
|
+
"array",
|
|
1092
|
+
"boolean"
|
|
1093
|
+
]
|
|
1094
|
+
},
|
|
1095
|
+
"required": [
|
|
1096
|
+
"content_type"
|
|
1097
|
+
]
|
|
1098
|
+
},
|
|
1099
|
+
"enable_layerwise_casting": {
|
|
1100
|
+
"type": "object",
|
|
1101
|
+
"properties": {
|
|
1102
|
+
"storage_dtype": {
|
|
1103
|
+
"type": "string"
|
|
1104
|
+
},
|
|
1105
|
+
"compute_dtype": {
|
|
1106
|
+
"type": "string"
|
|
1107
|
+
}
|
|
1108
|
+
},
|
|
1109
|
+
"required": [
|
|
1110
|
+
"storage_dtype",
|
|
1111
|
+
"compute_dtype"
|
|
1112
|
+
]
|
|
1113
|
+
},
|
|
1114
|
+
"group_offload": {
|
|
1115
|
+
"description": "Stream a model between system memory and the accelerator a group of layers at a time. Anything beyond the keys named here is passed through to apply_group_offloading, e.g. 'use_stream', 'num_blocks_per_group', 'low_cpu_mem_usage', 'offload_to_disk_path'.",
|
|
1116
|
+
"type": "object",
|
|
1117
|
+
"properties": {
|
|
1118
|
+
"onload_device": {
|
|
1119
|
+
"type": "string"
|
|
1120
|
+
},
|
|
1121
|
+
"offload_device": {
|
|
1122
|
+
"type": "string",
|
|
1123
|
+
"default": "cpu"
|
|
1124
|
+
},
|
|
1125
|
+
"offload_type": {
|
|
1126
|
+
"type": "string"
|
|
1127
|
+
}
|
|
1128
|
+
},
|
|
1129
|
+
"additionalProperties": true,
|
|
1130
|
+
"required": [
|
|
1131
|
+
"offload_type"
|
|
1132
|
+
]
|
|
1133
|
+
},
|
|
1134
|
+
"compile_config": {
|
|
1135
|
+
"description": "Compile the component with torch.compile once it is fully configured. Best paired with a pinned attention_backend - the per-call context manager forces recompiles.",
|
|
1136
|
+
"type": "object",
|
|
1137
|
+
"properties": {
|
|
1138
|
+
"mode": {
|
|
1139
|
+
"description": "torch.compile mode, e.g. 'default', 'reduce-overhead', 'max-autotune'.",
|
|
1140
|
+
"type": "string"
|
|
1141
|
+
},
|
|
1142
|
+
"fullgraph": {
|
|
1143
|
+
"description": "Require a single graph with no breaks. Fails fast on graph breaks instead of silently losing speedup.",
|
|
1144
|
+
"type": "boolean"
|
|
1145
|
+
},
|
|
1146
|
+
"dynamic": {
|
|
1147
|
+
"description": "Compile with dynamic shapes. Set true when resolutions or frame counts vary between runs to avoid recompiles.",
|
|
1148
|
+
"type": "boolean"
|
|
1149
|
+
},
|
|
1150
|
+
"repeated_blocks": {
|
|
1151
|
+
"description": "Compile only the model's repeated block classes (diffusers regional compilation) - near the same speedup as full compilation with a fraction of the cold-start cost.",
|
|
1152
|
+
"type": "boolean"
|
|
1153
|
+
}
|
|
1154
|
+
}
|
|
1155
|
+
}
|
|
1156
|
+
}
|
|
1157
|
+
}
|