diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
|
@@ -0,0 +1,2038 @@
|
|
|
1
|
+
# Workflow Guide
|
|
2
|
+
|
|
3
|
+
## How the catalog is organised
|
|
4
|
+
|
|
5
|
+
`workflows/` holds two trees, and which one a file is in says what it is for.
|
|
6
|
+
|
|
7
|
+
**`workflows/templates/`** teaches a pattern. One file per capability - a shape
|
|
8
|
+
(image to video, a multi-shot cut sequence), a mechanism (shared components,
|
|
9
|
+
sub-workflows, `pipeline_reference`, typed references), or a reference
|
|
10
|
+
convention (`prompt:`, `previous_result:`). These are what to read and copy.
|
|
11
|
+
Where several checkpoints run the same pattern through the same pipeline class,
|
|
12
|
+
one template carries them all and its `description` spells out the per-checkpoint
|
|
13
|
+
argument sets, so the variations travel with the file rather than in a document
|
|
14
|
+
that drifts from it. The `templates/ltx2/` and `templates/minimax/` subfolders
|
|
15
|
+
each hold a family whose members build on one baseline.
|
|
16
|
+
|
|
17
|
+
**`workflows/models/`** records a hardware fact: the quantization, offloading and
|
|
18
|
+
component placement that make one checkpoint fit a real card. That is knowledge
|
|
19
|
+
you cannot re-derive from a template, so it is kept runnable - but nobody learns
|
|
20
|
+
a pattern from the fifth one, so these stay out of the way. Each carries a
|
|
21
|
+
`configures` naming the template it is an instance of:
|
|
22
|
+
|
|
23
|
+
```json
|
|
24
|
+
{
|
|
25
|
+
"id": "flux-dev",
|
|
26
|
+
"description": "Text-to-image with FLUX.1 dev - the reference FLUX workflow.",
|
|
27
|
+
"configures": "templates/text-to-image",
|
|
28
|
+
"steps": [ ... ]
|
|
29
|
+
}
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
The distinction exists because a catalog entry that cannot say which of the two
|
|
33
|
+
it is leaves every reader - and every agent - to guess from the filename.
|
|
34
|
+
|
|
35
|
+
## Structure
|
|
36
|
+
|
|
37
|
+
Every workflow is a JSON file with an `id`, optional `variables`, and a list of `steps`:
|
|
38
|
+
|
|
39
|
+
```json
|
|
40
|
+
{
|
|
41
|
+
"id": "my_workflow",
|
|
42
|
+
"variables": {
|
|
43
|
+
"prompt": "default prompt text",
|
|
44
|
+
"steps": 25
|
|
45
|
+
},
|
|
46
|
+
"steps": [ ... ]
|
|
47
|
+
}
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
**Variables** define defaults that can be overridden from the command line:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
python -m dw.run my_workflow.json prompt="a cat" steps=50
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Variable names must be alphanumeric with underscores or hyphens.
|
|
57
|
+
|
|
58
|
+
## Step Types
|
|
59
|
+
|
|
60
|
+
Each step has a `name` and exactly one of four types:
|
|
61
|
+
|
|
62
|
+
### Pipeline Steps
|
|
63
|
+
|
|
64
|
+
Run a HuggingFace Diffusers model:
|
|
65
|
+
|
|
66
|
+
```json
|
|
67
|
+
{
|
|
68
|
+
"name": "generate",
|
|
69
|
+
"pipeline": {
|
|
70
|
+
"configuration": { "component_type": "FluxPipeline" },
|
|
71
|
+
"from_pretrained_arguments": {
|
|
72
|
+
"model_name": "black-forest-labs/FLUX.1-dev",
|
|
73
|
+
"torch_dtype": "torch.bfloat16"
|
|
74
|
+
},
|
|
75
|
+
"arguments": {
|
|
76
|
+
"prompt": "variable:prompt",
|
|
77
|
+
"num_inference_steps": 25
|
|
78
|
+
}
|
|
79
|
+
},
|
|
80
|
+
"result": { "content_type": "image/jpeg" }
|
|
81
|
+
}
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
### Pipeline Reference Steps
|
|
85
|
+
|
|
86
|
+
Re-run an already-loaded pipeline from an earlier step with a fresh set of arguments,
|
|
87
|
+
instead of loading the model again. This is how a two-pass technique like RF-Inversion
|
|
88
|
+
works: an `invert` step loads the pipeline, and a `main` step reuses it with the
|
|
89
|
+
inverted latents:
|
|
90
|
+
|
|
91
|
+
```json
|
|
92
|
+
{
|
|
93
|
+
"name": "main",
|
|
94
|
+
"pipeline_reference": {
|
|
95
|
+
"reference_name": "invert",
|
|
96
|
+
"arguments": {
|
|
97
|
+
"prompt": "variable:prompt",
|
|
98
|
+
"inverted_latents": "previous_result:invert.inverted_latents",
|
|
99
|
+
"image_latents": "previous_result:invert.image_latents"
|
|
100
|
+
}
|
|
101
|
+
},
|
|
102
|
+
"result": { "content_type": "image/jpeg" }
|
|
103
|
+
}
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
`reference_name` must name a step earlier in the same workflow that has a `pipeline`.
|
|
107
|
+
See [workflows/templates/community-pipeline.json](../workflows/templates/community-pipeline.json) for a full example.
|
|
108
|
+
|
|
109
|
+
### Task Steps
|
|
110
|
+
|
|
111
|
+
Run utility operations (image processing, QR codes, data gathering):
|
|
112
|
+
|
|
113
|
+
```json
|
|
114
|
+
{
|
|
115
|
+
"name": "preprocess",
|
|
116
|
+
"task": {
|
|
117
|
+
"command": "canny",
|
|
118
|
+
"arguments": {
|
|
119
|
+
"image": { "location": "https://example.com/photo.jpg" }
|
|
120
|
+
}
|
|
121
|
+
},
|
|
122
|
+
"result": { "content_type": "image/jpeg" }
|
|
123
|
+
}
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
A task can take `inputs` (a plain array) instead of `arguments`. Each array item becomes
|
|
127
|
+
its own iteration, the same way multiple `previous_result` values do. An item that is a
|
|
128
|
+
`previous_result:` reference becomes one iteration per result it names, and an object item
|
|
129
|
+
expands the way an `arguments` object would:
|
|
130
|
+
|
|
131
|
+
```json
|
|
132
|
+
{
|
|
133
|
+
"name": "prompts",
|
|
134
|
+
"task": {
|
|
135
|
+
"command": "gather_inputs",
|
|
136
|
+
"inputs": ["a marmot on a bicycle", "a bug driving a cycle"]
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
### Workflow Steps
|
|
142
|
+
|
|
143
|
+
Invoke another workflow file:
|
|
144
|
+
|
|
145
|
+
```json
|
|
146
|
+
{
|
|
147
|
+
"name": "expand",
|
|
148
|
+
"workflow": {
|
|
149
|
+
"path": "builtin:h3_context_ir.json",
|
|
150
|
+
"arguments": { "prompt": "variable:prompt" }
|
|
151
|
+
},
|
|
152
|
+
"result": { "content_type": "text/plain" }
|
|
153
|
+
}
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
`path` is read the way `run_workflow`'s `workflow_path` is: a catalog name as
|
|
157
|
+
`list_workflows` reports it (`templates/minimax/reference-to-video`), with or
|
|
158
|
+
without `.json`; a path relative to the file that names it (`../models/x.json`);
|
|
159
|
+
or `builtin:name.json` for the packaged fragments in `dw/workflows/`. A name
|
|
160
|
+
resolves beside the referencing file first, then against the run's own
|
|
161
|
+
`workflows/` directory, then against each read-only source the server lists -
|
|
162
|
+
so a stored template can be composed without copying it into the workspace. A
|
|
163
|
+
path that lands outside every source is refused, and one that resolves nowhere
|
|
164
|
+
is a validation error rather than a run that fails on its first step.
|
|
165
|
+
|
|
166
|
+
When the composing step declares a `result`, that is where the composed output
|
|
167
|
+
is written, once: the child's own last step does not save it a second time
|
|
168
|
+
under its own name. A composing step that declares no `result` (or one with no
|
|
169
|
+
`content_type`) leaves the saving to the child, as before. The child's other
|
|
170
|
+
steps write into the same run directory, with the composing step's name
|
|
171
|
+
leading their file names.
|
|
172
|
+
|
|
173
|
+
## Cross-Step Data Flow
|
|
174
|
+
|
|
175
|
+
### Variable References
|
|
176
|
+
|
|
177
|
+
Reference workflow variables with `variable:name`:
|
|
178
|
+
|
|
179
|
+
```json
|
|
180
|
+
"prompt": "variable:prompt"
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
A variable's declared value is both its default and its type — a value passed in is
|
|
184
|
+
converted to the type of the default, so declaring `25` and `"25"` are different things
|
|
185
|
+
(see the schema note under Variables). Declaring `null` opts out of that: the variable
|
|
186
|
+
becomes optional and untyped, taking whatever it is given and staying `null` when it is
|
|
187
|
+
given nothing.
|
|
188
|
+
|
|
189
|
+
```json
|
|
190
|
+
"variables": { "image": null }
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
A value the library already declares does not need a variable at all - see
|
|
194
|
+
[Constant References](#constant-references).
|
|
195
|
+
|
|
196
|
+
This is how a workflow exposes an argument a caller *may* pass without inventing a
|
|
197
|
+
sentinel for its absence — a sub-workflow that behaves differently when handed an image,
|
|
198
|
+
say. A caller can only set variables the workflow declares, so an optional argument still
|
|
199
|
+
has to be declared to be passable.
|
|
200
|
+
|
|
201
|
+
### Previous Result References
|
|
202
|
+
|
|
203
|
+
Pass output from one step to another with `previous_result:step_name`:
|
|
204
|
+
|
|
205
|
+
```json
|
|
206
|
+
{
|
|
207
|
+
"steps": [
|
|
208
|
+
{
|
|
209
|
+
"name": "preprocess",
|
|
210
|
+
"task": { "command": "canny", "arguments": { "image": { "location": "photo.jpg" } } }
|
|
211
|
+
},
|
|
212
|
+
{
|
|
213
|
+
"name": "generate",
|
|
214
|
+
"pipeline": {
|
|
215
|
+
"arguments": {
|
|
216
|
+
"control_image": "previous_result:preprocess",
|
|
217
|
+
"prompt": "a painting"
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
]
|
|
222
|
+
}
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
A reference is resolved wherever it appears in the arguments, not only at the top of
|
|
226
|
+
them - an argument holding a list or a nested object can reference a step too, which is
|
|
227
|
+
what lets a constructed object be
|
|
228
|
+
[built from an earlier step](#objects-built-from-an-earlier-step).
|
|
229
|
+
|
|
230
|
+
Multiple `previous_result` references create a **cartesian product**: if step A produces 4 images and step B produces 3 masks, a step referencing both will run 12 times.
|
|
231
|
+
|
|
232
|
+
A step whose result is a dict (a task returning several named outputs, or a pipeline
|
|
233
|
+
step that returns something like `inverted_latents`) can be referenced property by
|
|
234
|
+
property with `previous_result:step_name.property_name`:
|
|
235
|
+
|
|
236
|
+
```json
|
|
237
|
+
"inverted_latents": "previous_result:invert.inverted_latents"
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
### Media Arguments
|
|
241
|
+
|
|
242
|
+
Images and videos load automatically for arguments named `image`/`*_image` and
|
|
243
|
+
`video`/`*_video`. Any other argument - `mask`, `depth_map`, a controlnet's second
|
|
244
|
+
conditioning image - can load the same way with an explicit form that says what the
|
|
245
|
+
media is instead of relying on its argument name:
|
|
246
|
+
|
|
247
|
+
```json
|
|
248
|
+
"mask": { "media_type": "image", "location": "mask.png" }
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
`media_type` is `"image"` or `"video"`. `location` is a path relative to the workflow
|
|
252
|
+
file, or a URL, exactly like the plain `image`/`video` forms.
|
|
253
|
+
|
|
254
|
+
## Authoring a workflow from an agent
|
|
255
|
+
|
|
256
|
+
For an agent that has read the catalog (`list_workflows`), found nothing that
|
|
257
|
+
produces the shape it needs, and is about to write JSON. `get_schema` says what
|
|
258
|
+
is *well-formed*; this section says what the engine *does* with a well-formed
|
|
259
|
+
document, which is where a draft that validates still fails.
|
|
260
|
+
|
|
261
|
+
### References
|
|
262
|
+
|
|
263
|
+
An argument value is a reference when it begins with one of these prefixes.
|
|
264
|
+
Each resolves before the step runs. `variable:` and `previous_result:` names
|
|
265
|
+
are checked statically, so a bad one is a validation error at the path it
|
|
266
|
+
sits at; a `constant:`, `asset:`, `prompt:` or `output:` name in the
|
|
267
|
+
definition body resolves only when the step runs, and one that is missing
|
|
268
|
+
fails the run - unless it arrives in the `arguments` passed to
|
|
269
|
+
`validate_workflow`, which checks an `asset:`, `prompt:` or `output:` there
|
|
270
|
+
for existence.
|
|
271
|
+
|
|
272
|
+
- `variable:` — `variable:name` is the workflow's own `variables` entry,
|
|
273
|
+
overridden by the caller's `arguments`. A variable declared `null` is optional and untyped.
|
|
274
|
+
Schema validation runs before substitution, so a default must already be the
|
|
275
|
+
JSON type the field expects: `25`, not `"25"`.
|
|
276
|
+
- `previous_result:` — `previous_result:step_name` is the outputs of an earlier
|
|
277
|
+
step, named by that step's `name`. It iterates; see the cartesian rule
|
|
278
|
+
below. Validation checks it: a literal reference naming no earlier step is
|
|
279
|
+
an error with the JSON path it sits at, rather than a run-time failure
|
|
280
|
+
reached after everything before it has generated. A `.field` suffix
|
|
281
|
+
(`previous_result:invert.inverted_latents`) picks one field of a result that
|
|
282
|
+
is a dict, or a data attribute of a result object.
|
|
283
|
+
- `constant:` — `constant:module.path.NAME` is a value declared in Python,
|
|
284
|
+
read by import rather than copied into JSON. Anything callable is refused.
|
|
285
|
+
- `asset:` — `asset:name` is a file in the asset library. Rooted at the
|
|
286
|
+
library and confined to it, never resolved relative to the workflow file; a path that
|
|
287
|
+
escapes the library is rejected. `upload_asset` returns one of these names.
|
|
288
|
+
One name, three places it can sit: a bare string (`"image": "asset:x.png"`),
|
|
289
|
+
an element of a list, or the `location` of a media dict
|
|
290
|
+
(`{"location": "asset:x.png"}`, with or without `media_type`) - all resolve
|
|
291
|
+
to the same file.
|
|
292
|
+
- `output:` — `output:<workflow identity>/<run id>/<file>` is a file an earlier
|
|
293
|
+
run wrote, under the output root and confined to it. `latest` in the run-id position
|
|
294
|
+
picks the newest run that holds that file; `v<N>` picks the run the gallery labels
|
|
295
|
+
`v<N>` (`list_gallery`'s `version`), and only that run. A run id is not stable against
|
|
296
|
+
pruning: to depend on a generated file, promote it with `keep_output` and
|
|
297
|
+
reference the `asset:` name instead.
|
|
298
|
+
- `prompt:` — `prompt:name` or `prompt:folder/name` is a stored prompt's
|
|
299
|
+
`text`, rooted at the prompt library. That text may not itself begin with any of these
|
|
300
|
+
prefixes; the engine rejects such a prompt rather than resolving twice.
|
|
301
|
+
The library is also the worked-example shelf: a template's default prompt
|
|
302
|
+
is usually a `prompt:` reference, and the text behind it is a caption
|
|
303
|
+
written to whatever spec that model was trained on. Before writing a
|
|
304
|
+
prompt for a family, read the one that is already there —
|
|
305
|
+
`list_prompts(intended_model="ltx-2.5")` for the shelf,
|
|
306
|
+
`get_prompt("ltx2/fox_dawn_choir")` for a body. The listing leaves the
|
|
307
|
+
bodies out by default and reports each one's `text_chars`; asking for all
|
|
308
|
+
of them at once is more than a client will accept.
|
|
309
|
+
- `item:` — only inside a step that carries `for_each`: `item:` is the
|
|
310
|
+
entry the member was made for, `item:field` one field of an object entry,
|
|
311
|
+
spliced in whole whatever its type — a string, a number, a list of
|
|
312
|
+
references. See "One step per entry" below.
|
|
313
|
+
- `gather:` — `gather:shot` is the result of *every* member of the
|
|
314
|
+
`for_each` step `shot`, in list order, as one list. Inside a list it splices
|
|
315
|
+
into it. It is how a step downstream of a fan-out reads the whole group;
|
|
316
|
+
`previous_result:shot` naming a `for_each` step is an error that says so.
|
|
317
|
+
|
|
318
|
+
After a long inline run that is worth keeping, `get_job_workflow(job_id)`
|
|
319
|
+
returns the realized workflow — the definition with the arguments, seed and
|
|
320
|
+
prompts of that run pinned into it — and `save_workflow` gives it a name, so
|
|
321
|
+
the next run is by name rather than by pasting JSON again. `export_job(job_id)`
|
|
322
|
+
bundles the whole run (workflow, manifest, job row, the media on both sides)
|
|
323
|
+
into a directory on the server plus a zip URL, for a run worth committing or
|
|
324
|
+
handing to someone else.
|
|
325
|
+
|
|
326
|
+
A reference is resolved wherever it appears in the arguments, including inside
|
|
327
|
+
a nested object or list — not only at the top level. It is always the *whole*
|
|
328
|
+
value: `"variable:base_prompt"` resolves, `"variable:base_prompt, in fog"` asks
|
|
329
|
+
for a variable named `base_prompt, in fog` and fails the run. Nothing is
|
|
330
|
+
interpolated around a reference. To vary a fixed prompt across steps, write
|
|
331
|
+
each full prompt out, or put the shared text in a variable and let a step's
|
|
332
|
+
argument override it whole. A `variable:` reference that names nothing the
|
|
333
|
+
workflow declares is a validation error, not a warning: once a `variables`
|
|
334
|
+
block exists the engine refuses an undeclared reference, so it is a run that
|
|
335
|
+
cannot start.
|
|
336
|
+
|
|
337
|
+
When several steps share a block of text — a character's description and voice
|
|
338
|
+
repeated in every shot of a dialogue short — the answer is composition rather
|
|
339
|
+
than interpolation: write the shared text once as a variable and assemble each
|
|
340
|
+
step's prompt with a [`compose_text`](TASKS.md#compose_text) task, whose parts
|
|
341
|
+
are whole references joined in order. The shot then references the composed
|
|
342
|
+
result (`"prompt": "previous_result:shot_1_prompt"`), so changing the voice
|
|
343
|
+
changes it in every shot instead of in however many copies were made by hand.
|
|
344
|
+
|
|
345
|
+
### Types and escaping
|
|
346
|
+
|
|
347
|
+
Any key ending in `_type` or `_dtype`, or named `dtype`, has its string value
|
|
348
|
+
loaded as a Python object: `"FluxPipeline"` from `diffusers`, a dotted name
|
|
349
|
+
(`"torch.bfloat16"`, `"sdnq.SDNQConfig"`) by full module path. Wrapping a value
|
|
350
|
+
in braces keeps it a plain string — `"{nf4}"` is the string `nf4`. Getting this
|
|
351
|
+
wrong fails at load time, after validation has already passed, so a value that
|
|
352
|
+
is meant as text under one of those keys must be braced.
|
|
353
|
+
|
|
354
|
+
### What a variable is allowed to be
|
|
355
|
+
|
|
356
|
+
A model's own rule about a value belongs in the workflow, not in engine code
|
|
357
|
+
(CLAUDE.md) and not in a consumer's head. `variable_constraints` declares it
|
|
358
|
+
per variable, in the same field names a chain step's `frame_snap` uses:
|
|
359
|
+
|
|
360
|
+
```json
|
|
361
|
+
"variable_constraints": {
|
|
362
|
+
"num_frames": {
|
|
363
|
+
"modulus": 17,
|
|
364
|
+
"remainder": 5,
|
|
365
|
+
"min_frames": 124,
|
|
366
|
+
"max_frames": 345,
|
|
367
|
+
"snap": "up",
|
|
368
|
+
"reason": "the video VAE encodes 17 * n + 5 frames, and MiniMax-H3 generates between 5 and 15 seconds at 24 fps"
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
```
|
|
372
|
+
|
|
373
|
+
The value has to be `modulus * n + remainder` within `min_frames` to
|
|
374
|
+
`max_frames`. With `snap: "up"` an off-grid value is rounded to the next one
|
|
375
|
+
the rule accepts and the run *says so* - `130` becomes `141`, reported as a
|
|
376
|
+
warning at validation time and again in the job's `warnings`; without `snap`
|
|
377
|
+
it is refused. The bounds are checked against the value the run will use, so
|
|
378
|
+
they hold for the rounded number: on the rule above `108` is accepted (it
|
|
379
|
+
becomes `124`) and `346` is refused (it would become `362`).
|
|
380
|
+
|
|
381
|
+
Checked three times, for the reasons the task-argument domains are: in
|
|
382
|
+
`validation_errors`, so `POST /api/validate`, `validate_workflow` and the
|
|
383
|
+
pre-queue check all refuse a bad value at `arguments.<name>` or
|
|
384
|
+
`variables.<name>` for free; at run time before anything loads, which is the
|
|
385
|
+
backstop for a value the static pass cannot see (an inline workflow, a value
|
|
386
|
+
a parent passed down); and in the catalog, where `list_workflows` and
|
|
387
|
+
`get_workflow(variables_only=true)` report the rule beside the default - the
|
|
388
|
+
half that stops the next caller picking a number the model refuses.
|
|
389
|
+
|
|
390
|
+
State the rule once. Where a template both declares a constraint and snaps a
|
|
391
|
+
chain, the chain's `frame_snap` names it rather than repeating the numbers:
|
|
392
|
+
|
|
393
|
+
```json
|
|
394
|
+
"frame_snap": "constraint:num_frames"
|
|
395
|
+
```
|
|
396
|
+
|
|
397
|
+
Two limits, both accepted. A constraint cannot express a bound that depends
|
|
398
|
+
on another variable (a maximum that is `fps * seconds` where a template
|
|
399
|
+
exposes `fps`), and it reaches a top-level variable only - not a field inside
|
|
400
|
+
a list entry, so a `for_each` template whose entries each carry their own
|
|
401
|
+
`num_frames` is unconstrained and relies on the run-time check.
|
|
402
|
+
|
|
403
|
+
### A workflow takes only the keys the engine reads
|
|
404
|
+
|
|
405
|
+
The workflow object itself, `step`, `task`, `workflow`,
|
|
406
|
+
`pipeline_reference` and `result` are closed: a property the engine does not
|
|
407
|
+
read is a validation error naming the object and the key, not a warning.
|
|
408
|
+
There is no `when`, no `retry`, no `select` - if a draft reaches for one, the
|
|
409
|
+
shape it wants is a different arrangement of steps, not a flag. The error
|
|
410
|
+
exists because a plausible invented key used to validate cleanly and then do
|
|
411
|
+
nothing, so the expensive work ran with the input silently having had no
|
|
412
|
+
effect - a mistyped `sedd` left the run unseeded while validation advised
|
|
413
|
+
setting a seed, and a mistyped `subfoldr` put the deliverable at the run
|
|
414
|
+
root rather than in `final/`.
|
|
415
|
+
|
|
416
|
+
`pipeline` is closed to the same rule with one opening: any key whose value
|
|
417
|
+
is a *component definition* - an object carrying `from_pretrained_arguments` -
|
|
418
|
+
names a component to load, because diffusers grows component names faster
|
|
419
|
+
than the schema does (`latent_upsampler`, `prompt_enhancer` and `processor`
|
|
420
|
+
all appear that way in shipped templates). A pipeline key that is not one of
|
|
421
|
+
those is refused, which is what catches `pipeline_type` or `model_name`
|
|
422
|
+
written a level too high. `from_pretrained_arguments` stays open - it passes
|
|
423
|
+
its keys through to `from_pretrained`.
|
|
424
|
+
|
|
425
|
+
### Where a workflow may read and reach
|
|
426
|
+
|
|
427
|
+
An argument that names a *location* is confined, untrusted (the default):
|
|
428
|
+
|
|
429
|
+
- a path must resolve inside the workflow's own directory, the asset
|
|
430
|
+
libraries, or the output root. An absolute path elsewhere is refused
|
|
431
|
+
whether or not it exists. The remedy is `upload_asset` (or `keep_output`)
|
|
432
|
+
and an `asset:` reference - which is what those exist for.
|
|
433
|
+
- a `glob` is confined the same way, and each match re-checked.
|
|
434
|
+
- an `http(s)` URL may not resolve to an address inside the deployment -
|
|
435
|
+
loopback, link-local, private ranges.
|
|
436
|
+
- `remote_text_encoder.url` is https-only, and only a HuggingFace host is
|
|
437
|
+
sent this machine's token.
|
|
438
|
+
- `model_name` must be a Hub repo id, or a path inside one of those roots.
|
|
439
|
+
|
|
440
|
+
All of it is reported by `validate_workflow`, before anything is queued, so
|
|
441
|
+
a draft that names a file the server may not read costs nothing to find out.
|
|
442
|
+
`get_server_info`'s `trust_workflows` says which posture is in force.
|
|
443
|
+
|
|
444
|
+
### Remote code is refused by default
|
|
445
|
+
|
|
446
|
+
A server started without `--trust-workflows` refuses any
|
|
447
|
+
`from_pretrained_arguments` that sets `trust_remote_code` or
|
|
448
|
+
`custom_pipeline`, at load time, after validation has passed. Use a
|
|
449
|
+
pipeline diffusers ships: no bundled catalog entry carries either key, and a
|
|
450
|
+
workflow that does runs only on a server whose operator turned trust on,
|
|
451
|
+
which `get_server_info` reports as `trust_workflows`.
|
|
452
|
+
|
|
453
|
+
### Several `previous_result` references multiply
|
|
454
|
+
|
|
455
|
+
When one step carries two or more `previous_result` references, the engine runs
|
|
456
|
+
that step once for every combination — a cartesian product. Four images and
|
|
457
|
+
three masks is twelve iterations, not three pairs. Past 10000 combinations the
|
|
458
|
+
run is refused outright.
|
|
459
|
+
|
|
460
|
+
This is deliberate: it is how one prompt fans out over a set. The consequence
|
|
461
|
+
is that a *pairing* — shot *i* with speaker *i*, prompt *i* with portrait *i* —
|
|
462
|
+
cannot be expressed with two references on one step. Write it as one step per
|
|
463
|
+
pair, each referencing exactly the two things it pairs, or gather the pairs
|
|
464
|
+
upstream so each is a single result. A step that seems to need a "zip" is the
|
|
465
|
+
signal to restructure the workflow, not to add another reference.
|
|
466
|
+
|
|
467
|
+
### One step per entry: `for_each`
|
|
468
|
+
|
|
469
|
+
A step that carries `for_each` runs once per entry of a list — a shot per
|
|
470
|
+
entry of `shots` — and the list is a variable the caller supplies, so a
|
|
471
|
+
six-shot episode is an argument rather than a different file.
|
|
472
|
+
|
|
473
|
+
```json
|
|
474
|
+
{
|
|
475
|
+
"name": "shot",
|
|
476
|
+
"for_each": "variable:shots",
|
|
477
|
+
"pipeline": {
|
|
478
|
+
"arguments": {
|
|
479
|
+
"prompt": "item:prompt",
|
|
480
|
+
"references": "item:references"
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
```
|
|
485
|
+
|
|
486
|
+
with
|
|
487
|
+
|
|
488
|
+
```json
|
|
489
|
+
"shots": [
|
|
490
|
+
{ "name": "wide_open", "prompt": "the band walks on, wide",
|
|
491
|
+
"references": [{ "reference_type": "…", "from_previous_result": "draw_singer" }] },
|
|
492
|
+
{ "name": "closeup", "prompt": "closeup on the singer",
|
|
493
|
+
"references": [{ "reference_type": "…", "from_previous_result": "draw_singer" }] }
|
|
494
|
+
]
|
|
495
|
+
```
|
|
496
|
+
|
|
497
|
+
and downstream
|
|
498
|
+
|
|
499
|
+
```json
|
|
500
|
+
{ "name": "edit",
|
|
501
|
+
"task": { "command": "concat_videos", "arguments": { "videos": "gather:shot" } } }
|
|
502
|
+
```
|
|
503
|
+
|
|
504
|
+
Before the run starts, the engine replaces the `for_each` step with one
|
|
505
|
+
ordinary step per entry, named `shot@wide_open`, `shot@closeup` — the
|
|
506
|
+
entry's `name`, or its index for an entry without one. Those are the names
|
|
507
|
+
the manifest, the job's events and the gallery show, and `@` is reserved
|
|
508
|
+
for them: a hand-written step name may not contain it. An entry's `name`
|
|
509
|
+
must be unique in its list and match `^[a-zA-Z_][a-zA-Z0-9_-]*$`. Give
|
|
510
|
+
entries names: the step cache keys on the member name, so a shot inserted
|
|
511
|
+
in the middle of a named list leaves every other shot cached, while an
|
|
512
|
+
indexed list shifts every later shot onto a different entry and regenerates
|
|
513
|
+
it.
|
|
514
|
+
|
|
515
|
+
`item:field` is the whole value of that field, so an entry can carry
|
|
516
|
+
anything a step argument can — including a `references` list whose length
|
|
517
|
+
differs by shot, with `from_previous_result` and `asset:` strings inside
|
|
518
|
+
it. Nothing is interpolated: `"item:prompt"` is the field, `"shot: item:prompt"`
|
|
519
|
+
is a literal string.
|
|
520
|
+
|
|
521
|
+
An entry may name another variable: `"from_file": "variable:character_a_voice"`
|
|
522
|
+
inside a `references` entry is that variable's value by the time the member
|
|
523
|
+
exists, so one variable sets a voice in every shot the character speaks in
|
|
524
|
+
and a caller who supplies the list still writes `variable:` for the parts the
|
|
525
|
+
template fixes. Those references are resolved before anything in the entry is
|
|
526
|
+
loaded, and an undeclared one is a validation error at the entry's path
|
|
527
|
+
(`arguments.shots[2].references[1].from_file` when the list is yours,
|
|
528
|
+
`variables.shots[...]` when it is the template's). A value may not reference
|
|
529
|
+
itself, directly or through another variable.
|
|
530
|
+
|
|
531
|
+
Two `for_each` steps over the *same* list are paired by key — the entry's
|
|
532
|
+
`name`, or its index for an entry without one: inside `shot@closeup`, a
|
|
533
|
+
reference to another `for_each` step `slice` over the same `shots` list
|
|
534
|
+
resolves to `slice@closeup`. That is how a shot reads the audio
|
|
535
|
+
slice cut for it when slicing and generating are two steps. It is the one
|
|
536
|
+
pairing the engine has; `for_each` runs over exactly one list, and there is
|
|
537
|
+
no zip and no loop index.
|
|
538
|
+
|
|
539
|
+
Limits: a list has at most 32 entries, and an empty list is a validation
|
|
540
|
+
error — the step would run nothing. Validation realizes a `constant:`
|
|
541
|
+
default before checking it, so a list defaulted to a constant validates the
|
|
542
|
+
same way it will run. `release_pipeline` on a `for_each`
|
|
543
|
+
step releases after the *last* member. Each entry is a full generation, so
|
|
544
|
+
quote the cost before running a list-driven workflow: the listing's `lists`
|
|
545
|
+
block names the fields an entry takes and the steps over it, and its `cost`
|
|
546
|
+
carries `per_entry` once one entry has been measured. `validate_workflow`
|
|
547
|
+
with your `arguments` answers with a `plan` whose `estimate` already does
|
|
548
|
+
that arithmetic (`basis: per_entry`); without `per_entry` it extrapolates
|
|
549
|
+
the stored total linearly over your list (`basis: derived` - an estimate
|
|
550
|
+
rather than a measurement) and reports the stored total unchanged only
|
|
551
|
+
when your list is the one it was measured with (`basis: catalog`). Ahead of
|
|
552
|
+
all of those it quotes this box's own finished runs of the shape you are
|
|
553
|
+
about to run when it has any (`basis: observed`, with `runs` saying how
|
|
554
|
+
many) - quote the plan's figure and say which basis it has. An
|
|
555
|
+
entry key no step reads is a validation warning at the entry's path, so a
|
|
556
|
+
misspelt field is caught before the run. Then
|
|
557
|
+
`validate_workflow` with the
|
|
558
|
+
`arguments` you will run with: it expands your list, not the template's
|
|
559
|
+
default, resolves the variables your entries name, and reports a duplicate
|
|
560
|
+
name or a missing field at the entry's path.
|
|
561
|
+
|
|
562
|
+
Every error carries a path in the file you wrote, not in the expanded step
|
|
563
|
+
list: a bad reference inside a member is reported at the `for_each` step's
|
|
564
|
+
own path, with the member it failed in named in the message.
|
|
565
|
+
|
|
566
|
+
`templates/minimax/dialogue-short` and `templates/minimax/music-video` are
|
|
567
|
+
this shape: each takes one `shots` list, and `get_workflow` on either shows
|
|
568
|
+
the entry an item needs.
|
|
569
|
+
|
|
570
|
+
### The loop
|
|
571
|
+
|
|
572
|
+
1. `validate_workflow` — free and instant. It reports every schema error at
|
|
573
|
+
once, each with the JSON path it sits at, plus warnings for argument names
|
|
574
|
+
that do not appear in the real pipeline signature. It also catches a
|
|
575
|
+
`previous_result:` (or `from_previous_result`) that names no earlier step,
|
|
576
|
+
which is what renaming a step half way through leaves behind. Pass the
|
|
577
|
+
`arguments` you are going to run with as well: a name the workflow no
|
|
578
|
+
longer declares, a value that will not coerce to the declared type, and an
|
|
579
|
+
`asset:`, `prompt:` or `output:` reference that names nothing in this
|
|
580
|
+
workspace each come back at `arguments.<name>`, for free, instead of after
|
|
581
|
+
the model has loaded. Without them the verdict is about the stored
|
|
582
|
+
definition and its stock defaults - `checked_arguments` in the answer says
|
|
583
|
+
which it was.
|
|
584
|
+
2. Fix everything reported, including the warnings: a passing validation does
|
|
585
|
+
not mean the pipeline accepts the arguments, and a typo against a real
|
|
586
|
+
`__call__` shows up only as one of those warnings. The server only computes
|
|
587
|
+
signature warnings for a schema-valid draft — while schema errors remain it
|
|
588
|
+
returns `warnings: []`, so validate again after fixing them to see the
|
|
589
|
+
warnings.
|
|
590
|
+
3. `save_workflow` — validates again on the way in and returns the catalog
|
|
591
|
+
metadata the saved draft will carry.
|
|
592
|
+
4. `run_workflow` with `acknowledged_cost=true`, after telling the user what it
|
|
593
|
+
costs. Without the acknowledgement the call is refused. The figure to tell
|
|
594
|
+
them is the `plan` on the validate answer - `estimate.minutes` with its
|
|
595
|
+
`basis`, and every `downloads_required` entry named as its own line item,
|
|
596
|
+
since weights not on this box are minutes and gigabytes the cost block
|
|
597
|
+
never counted. `basis` says where the figure came from, and that is what
|
|
598
|
+
decides how to quote it: `observed` is this box's own finished runs of
|
|
599
|
+
this shape (the cold median over `runs` of them, preferred over any
|
|
600
|
+
curated figure) - "about N minutes, measured over M runs"; `per_entry` is
|
|
601
|
+
a measured per-entry rate re-priced for your list; `catalog` is a
|
|
602
|
+
measured total for a run whose lists are the ones it was measured with;
|
|
603
|
+
`derived` is that total extrapolated over a list you changed the length
|
|
604
|
+
of - say it is an estimate; `other_device` is a figure from another
|
|
605
|
+
accelerator - say so too; `unknown` is no figure at all. `gb` on a
|
|
606
|
+
`downloads_required` entry is null when the hub could not be asked, and
|
|
607
|
+
`steps`/`list_entries` say how many members the list actually produced.
|
|
608
|
+
Then pass that plan back:
|
|
609
|
+
`acknowledged_cost={"fingerprint": plan.fingerprint, "minutes":
|
|
610
|
+
plan.estimate.minutes, "downloads": [...]}` - the server refuses with 409
|
|
611
|
+
if the run's shape changed since the quote, and the refusal carries the
|
|
612
|
+
new plan to quote from. `true` is for a plan that was null. When `basis`
|
|
613
|
+
is `unknown`: a workflow you wrote
|
|
614
|
+
or copied carries no `cost` of its own, but the pipeline inside it usually
|
|
615
|
+
does: `list_workflows(include_models=true)` finds the `models/` entry that
|
|
616
|
+
loads the same checkpoint, and its per-image figure times the number of
|
|
617
|
+
images is the number to quote. Say "a few minutes" only when no entry with
|
|
618
|
+
that pipeline has been measured.
|
|
619
|
+
5. `wait_for_job` rather than a polling loop; call it again if it returns
|
|
620
|
+
`still_running: true`. One call blocks for at most 55 seconds whatever
|
|
621
|
+
`timeout_seconds` says, so a minutes-long render takes several - the
|
|
622
|
+
reply's `timeout_capped` and `waited_seconds` say which happened. A
|
|
623
|
+
running job's `progress` carries the step being run; the phase
|
|
624
|
+
(`loading`, `generating`, `decoding`, `saving`) with the model it names in
|
|
625
|
+
`phase_detail`; `seconds_in_phase`, time spent in that phase; and
|
|
626
|
+
`seconds_since_event`, time since the last progress event - a number that
|
|
627
|
+
climbs while `denoise_step` stays put is the "nothing is happening" read.
|
|
628
|
+
It also carries `denoise_step`/`denoise_total_steps`,
|
|
629
|
+
null until the denoise loop starts; judge a slow run against a stuck one
|
|
630
|
+
by whether `denoise_step` has moved since a poll minutes ago, not by
|
|
631
|
+
silence past a fixed threshold. A null `denoise_step` under `generating`
|
|
632
|
+
is the pipeline's lead-in - encoding the prompt and every reference -
|
|
633
|
+
which emits nothing and can run for many minutes when a video reference
|
|
634
|
+
is among them; gaps between denoise steps are uneven too where a
|
|
635
|
+
transformer block cache is configured. Both are normal, and the model
|
|
636
|
+
family's own skill carries the measured figures. `denoise_total_steps` is
|
|
637
|
+
the schedule the pipeline actually runs, which is not always the
|
|
638
|
+
`num_inference_steps` that was asked for: MiniMax H3's scheduler counts
|
|
639
|
+
sigma grid points including the terminal zero, so it runs N-1 model
|
|
640
|
+
evaluations for N (9 reports 8, 20 reports 19) - the vendor's convention,
|
|
641
|
+
not a dropped step; raising the number still buys the steps it looks
|
|
642
|
+
like it does.
|
|
643
|
+
6. `get_output_image` to look at what was actually made, and say whether it
|
|
644
|
+
answers the request. Nothing before this step establishes that it does.
|
|
645
|
+
`get_output_frames` looks at a video and `get_output_audio` listens to a
|
|
646
|
+
soundtrack.
|
|
647
|
+
|
|
648
|
+
To confirm the words a clip speaks - a text-only client can't consume the
|
|
649
|
+
`AudioContent` block `get_output_audio` returns - transcribe it instead.
|
|
650
|
+
`validate_workflow(name="templates/transcribe-audio",
|
|
651
|
+
arguments={"input_audio": "output:<name>"})` first (free; it takes an
|
|
652
|
+
audio file or a video's muxed soundtrack directly), then
|
|
653
|
+
`run_workflow(..., acknowledged_cost={"fingerprint": ..., "minutes": ...,
|
|
654
|
+
"downloads": [...]})` bound to that plan with `wait_seconds=55`, then
|
|
655
|
+
`get_output_text` on the result, and `delete_output(job_id=...)` the
|
|
656
|
+
scratch run afterward. This workflow's plan comes back
|
|
657
|
+
`basis: "unknown"` with `minutes: null` - nothing is curated or observed
|
|
658
|
+
for it - so quote what it actually takes rather than the plan: seconds,
|
|
659
|
+
not minutes (a few seconds per clip in practice). Four calls and a short
|
|
660
|
+
wait, not a GPU-spending read tool - keep the normal queue rather than
|
|
661
|
+
adding one.
|
|
662
|
+
7. Getting the files to the user's machine. `download_output` and `export_job`
|
|
663
|
+
write on the machine running `dw.serve`, which over a remote `--mcp`
|
|
664
|
+
endpoint is the GPU box. The last mile of every deliverable is the `url`
|
|
665
|
+
each `list_gallery` entry carries (or `export_job`'s `zip_url`), fetched
|
|
666
|
+
with the same bearer token the MCP connection uses:
|
|
667
|
+
|
|
668
|
+
curl -H "Authorization: Bearer $DW_API_TOKEN" \
|
|
669
|
+
-o exports/still.png "http://<box>:8765/outputs/ltx2/Gyre/20260910-.../still.png"
|
|
670
|
+
|
|
671
|
+
Put the result under `exports/` in the session's working directory - it is
|
|
672
|
+
the user's deliverable, not a temporary file.
|
|
673
|
+
|
|
674
|
+
### Keeping a set consistent
|
|
675
|
+
|
|
676
|
+
"Four pictures of the same thing" is the commonest shape a request takes that
|
|
677
|
+
the catalog does not name directly, and what "the same" means decides the
|
|
678
|
+
workflow.
|
|
679
|
+
|
|
680
|
+
- **The same style, different subjects or scenes.** One prompt per picture,
|
|
681
|
+
the same `seed` on the workflow, and a shared style phrase in every prompt.
|
|
682
|
+
A shared seed does not make the pictures alike; it makes the run
|
|
683
|
+
reproducible. Consistency here comes from the prompts.
|
|
684
|
+
- **The same object, differing in one stated way** - four spoons identical
|
|
685
|
+
but for colour, one mug in four glazes, a product in each colourway.
|
|
686
|
+
Generate the object *once*, then run an image-edit pass per variant with
|
|
687
|
+
the base step's result as its `image` and an instruction that names only
|
|
688
|
+
the change ("make the mug red"). Separate generations, seeded or not, draw
|
|
689
|
+
a different object every time; an edit holds everything the instruction
|
|
690
|
+
does not mention. `templates/consistent-set.json` is this shape.
|
|
691
|
+
- **The same character in different situations.** A reference rather than an
|
|
692
|
+
edit: an identity-referencing pipeline or IP-Adapter conditioned on one
|
|
693
|
+
portrait, used by every picture (`templates/ip-adapter.json`,
|
|
694
|
+
`templates/multi-image-reference.json`, and for video the MiniMax
|
|
695
|
+
`reference-to-video` and `dialogue-short` templates). The `identity-referenced`
|
|
696
|
+
trait in the listing marks the workflows that take one.
|
|
697
|
+
|
|
698
|
+
### Saying which output is the deliverable
|
|
699
|
+
|
|
700
|
+
A run writes everything into one directory, so a finished episode sits
|
|
701
|
+
beside the twenty scratch files that went into it. A step's `result` block
|
|
702
|
+
can name a subfolder of the run directory for its files:
|
|
703
|
+
|
|
704
|
+
```json
|
|
705
|
+
"result": { "content_type": "video/mp4", "subfolder": "final" }
|
|
706
|
+
```
|
|
707
|
+
|
|
708
|
+
The convention is two names: `final` for a step whose output the user will
|
|
709
|
+
be shown, `intermediate` for everything else. The engine treats no name
|
|
710
|
+
specially and applies no default - a step that says nothing writes to the
|
|
711
|
+
run's root as it always has - but the gallery, `get_job` and `list_gallery`
|
|
712
|
+
all carry the value, so a consumer that follows the convention can tell the
|
|
713
|
+
deliverable from the scratch without knowing the workflow. Mark every saving
|
|
714
|
+
step of a multi-step workflow; a one-step workflow needs nothing.
|
|
715
|
+
|
|
716
|
+
The shipped templates follow it: every template with two or more saving
|
|
717
|
+
steps marks each one, so a workflow copied from a template starts with the
|
|
718
|
+
roles in place.
|
|
719
|
+
|
|
720
|
+
The value is a relative path of any depth (`shots/act-1`), may be a
|
|
721
|
+
`variable:` or, inside a `for_each` step, an `item:` reference, and follows
|
|
722
|
+
the `output:` segment rule - each segment starts with a letter, digit or
|
|
723
|
+
underscore; `..`, a backslash and a leading `.` are refused - so every
|
|
724
|
+
subfolder written is one a later workflow can name:
|
|
725
|
+
`output:dialogue-short/latest/final/episode.mp4`. A bad value is a
|
|
726
|
+
validation error at its JSON path. `file_base_name` is a name, not a path:
|
|
727
|
+
a separator there is refused, and `subfolder` is the way to place a file. It
|
|
728
|
+
replaces the name the engine would derive from the workflow and step rather
|
|
729
|
+
than prefixing it, so `"file_base_name": "episode"` in a `final` subfolder
|
|
730
|
+
writes `final/episode-0.0.mp4` - name each step that sets one differently, or
|
|
731
|
+
the second collides and picks up a `-2`.
|
|
732
|
+
|
|
733
|
+
A step that saves nothing and which no later step reads does not run at
|
|
734
|
+
all: the engine drops it before the first step executes and warns once per
|
|
735
|
+
dropped step. That is how a template whose portraits can be supplied as
|
|
736
|
+
`asset:` files stops paying for the steps that would have drawn them. It
|
|
737
|
+
follows from what the definition says, never from a value produced during
|
|
738
|
+
the run, so it is decided at validate time too - the `plan` a validate call
|
|
739
|
+
answers with counts only the steps that will run and lists the rest under
|
|
740
|
+
`elided_steps`. Four things keep a step: a `result` with a `content_type`
|
|
741
|
+
and `save` not `false`, being the last step, being read by a later step
|
|
742
|
+
(`previous_result:`, `gather:`, a `pipeline_reference`, a shared component),
|
|
743
|
+
or being read by a step that is itself kept - elision is transitive. If a
|
|
744
|
+
step you meant to run is named in the warnings, a reference to it is
|
|
745
|
+
misspelled somewhere later or it needs a `result`.
|
|
746
|
+
|
|
747
|
+
### Composing a stored workflow
|
|
748
|
+
|
|
749
|
+
A step with a `workflow` block runs another workflow as one step of this one,
|
|
750
|
+
with `arguments` handed down as that workflow's variables. Its `path` is read
|
|
751
|
+
the way `run_workflow`'s `workflow_path` is - a catalog name from
|
|
752
|
+
`list_workflows`, with or without `.json`, a path relative to the file that
|
|
753
|
+
names it, or `builtin:name.json` - and resolves beside the referencing file
|
|
754
|
+
first, then in this workspace's `workflows/`, then in each read-only source
|
|
755
|
+
the server lists. A stored template is composed by its catalog name; copying
|
|
756
|
+
it into the workspace to reach it is no longer necessary, and a copy silently
|
|
757
|
+
stops tracking the original.
|
|
758
|
+
|
|
759
|
+
Declare a `result` on the composing step and the composed output is saved
|
|
760
|
+
there, once, under that step's name and subfolder - the composed workflow's
|
|
761
|
+
own last step does not write a second copy. Its other steps write into the
|
|
762
|
+
same run directory, prefixed with the composing step's name.
|
|
763
|
+
|
|
764
|
+
`validate_workflow` resolves the path, so a name that reaches nothing is an
|
|
765
|
+
error at `steps[N].workflow.path` before anything is queued; it also validates
|
|
766
|
+
the workflow named, refuses a composition cycle, and warns about an argument
|
|
767
|
+
the composed workflow declares no variable for.
|
|
768
|
+
|
|
769
|
+
### Being found next time
|
|
770
|
+
|
|
771
|
+
The catalog derives each entry's `shape` — one of `image`, `image-set`,
|
|
772
|
+
`image-edit`, `shot`, `sequence`, `audio`, `text`, `utility` — and its `traits`
|
|
773
|
+
(`has-audio`, `chained`, `image-conditioned`, `identity-referenced`,
|
|
774
|
+
`needs-input-media`, `composes-workflows`) from the structure of the
|
|
775
|
+
definition, and its `summary` from the first sentence of `description`.
|
|
776
|
+
|
|
777
|
+
So write that first sentence to say what the workflow *makes* and what it
|
|
778
|
+
*needs supplied*, in under 120 characters — "H3 video with audio between two
|
|
779
|
+
supplied stills" — rather than what technique it demonstrates. A first sentence
|
|
780
|
+
longer than that is truncated with an ellipsis in every listing.
|
|
781
|
+
|
|
782
|
+
Declare `shape`, `traits` or `summary` at the top level only when derivation
|
|
783
|
+
gets it wrong; a declaration that merely repeats the derivation is noise that
|
|
784
|
+
rots when the rules change, and the repo's catalog tests refuse it. `cost` is
|
|
785
|
+
never derived — leave it absent until a run has been measured. That makes it
|
|
786
|
+
the catalog's verified marker as well: an entry carrying `cost` has been run
|
|
787
|
+
to completion on the device it names, and one without has only been authored
|
|
788
|
+
— its description may still say what it has not been able to check (VRAM at
|
|
789
|
+
a size, whether a format carries what the pipeline returns), and the first
|
|
790
|
+
run is the one that finds out.
|
|
791
|
+
|
|
792
|
+
`cost_drivers` is the other half of saying what a workflow costs, and it *is*
|
|
793
|
+
for derivation: the variables that move the wall clock — a frame count, a
|
|
794
|
+
step count, a segment count, the list a `for_each` runs over — never a prompt
|
|
795
|
+
or a seed. The server buckets its own finished runs by those values and
|
|
796
|
+
reports the result as `observed` beside the curated `cost`, so a 345-frame
|
|
797
|
+
run never informs a 124-frame figure and a list driver buckets on its length.
|
|
798
|
+
Declaring none is not neutral: the figure then falls back to runs that
|
|
799
|
+
overrode nothing at all, which most real runs do, so a measured workflow with
|
|
800
|
+
no drivers keeps answering "unknown". Each name must be a variable the
|
|
801
|
+
workflow declares — `tests/test_observed_cost.py` sweeps the catalog for one
|
|
802
|
+
that is not, since a driver bucketing on nothing looks exactly like a driver
|
|
803
|
+
that works.
|
|
804
|
+
|
|
805
|
+
## Assessing a run's output
|
|
806
|
+
|
|
807
|
+
A cut joined from shots can succeed and still be wrong at a seam, and the
|
|
808
|
+
whole-file numbers `get_gallery_metadata` reports cannot see inside a join.
|
|
809
|
+
`assess_output(name)` (`GET /api/gallery/{name}/assess`) measures that
|
|
810
|
+
file on the server. It decodes the file once, runs every assessment probe
|
|
811
|
+
that applies to it (`analyze_shots`, `analyze_seams`, `analyze_sync_drift`),
|
|
812
|
+
and returns where to look. It queues nothing: the probes use only the CPU
|
|
813
|
+
and run in the server process, beside whatever job holds the GPU. `name` is
|
|
814
|
+
a gallery name, `output:` or `asset:`. The shot boundaries come from what
|
|
815
|
+
the file's run recorded: the run manifest for an output, and the sidecar
|
|
816
|
+
`keep_output` wrote for an asset.
|
|
817
|
+
|
|
818
|
+
A last shot's `num_samples` a few dozen samples off `round(num_frames *
|
|
819
|
+
sample_rate / fps)` is expected, not a finding - see `pair_audio` in the tasks guide's
|
|
820
|
+
Video Processing section for why.
|
|
821
|
+
|
|
822
|
+
**Procedure.**
|
|
823
|
+
|
|
824
|
+
1. After `wait_for_job`, call `assess_output` on the deliverable, which is
|
|
825
|
+
the file under `final/`. `get_gallery_metadata` points at the tool
|
|
826
|
+
whenever `media.shots` is set.
|
|
827
|
+
2. Read `findings`. When the list is empty, no rule crossed its threshold,
|
|
828
|
+
which is a reason to listen less closely, not a pass.
|
|
829
|
+
3. Drill into each finding at the place its `at` names. For a seam, use
|
|
830
|
+
`get_output_frames(name, seams=[n])` to see it and
|
|
831
|
+
`get_output_audio(name, start, duration)` to hear the second around it.
|
|
832
|
+
For a shot, look at that shot's span. Judge it against the request.
|
|
833
|
+
4. Fix what you confirmed (see the table below), then assess the new cut.
|
|
834
|
+
|
|
835
|
+
Pass `probe="analyze_seams"` (or either of the other two probe names) to
|
|
836
|
+
get that one probe's full body: every seam's or shot's measurements, not
|
|
837
|
+
just the ones that crossed a rule. `detail=true` adds every applicable
|
|
838
|
+
probe's full body under `probes`. An unknown probe is refused before
|
|
839
|
+
anything is read, and the error names the three probes.
|
|
840
|
+
|
|
841
|
+
**The answer.**
|
|
842
|
+
|
|
843
|
+
| Field | What it holds |
|
|
844
|
+
| --- | --- |
|
|
845
|
+
| `findings` | Every rule a measurement crossed, each `{rule, severity, at, value, threshold, says}`. `severity` is `warn` or `info`. `at` names the shot or seam. |
|
|
846
|
+
| `rules_applied` | The rules that were checked, so an empty `findings` list says which checks came back clean. |
|
|
847
|
+
| `rules_skipped` | `{probe, rule, reason}` for each rule that could not be measured on this file. |
|
|
848
|
+
| `not_applicable` | `{probe: why}` for each probe that does not apply to this file. A still has no shots, seams or soundtrack. A mute file has no levels. A file with no recorded shots has no seams. |
|
|
849
|
+
| `shots_source` | Where the boundaries came from: `manifest` (the run's manifest, or an asset's sidecar), or `none`. |
|
|
850
|
+
|
|
851
|
+
The thresholds live in one table, `dw/assessment_rules.py`:
|
|
852
|
+
|
|
853
|
+
| Rule | Probe | Fires when |
|
|
854
|
+
| --- | --- | --- |
|
|
855
|
+
| `shot_level_spread` | `analyze_shots` | the shots' RMS levels span 6 dB or more |
|
|
856
|
+
| `seam_level_step` | `analyze_seams` | the shots either side of a seam differ by more than 3 dB |
|
|
857
|
+
| `seam_click` | `analyze_seams` | the join peaks more than 12 dB above the audio either side |
|
|
858
|
+
| `seam_hole` | `analyze_seams` | the join's floor drops below -50 dBFS while both sides are voiced (above -30 dBFS) |
|
|
859
|
+
| `seam_frame_jump` | `analyze_seams` | the picture changes more than 25x as much across the seam as inside either shot (`info`, and skipped at a shot marked `hard_cut: true`) |
|
|
860
|
+
| `sync_drift` | `analyze_sync_drift` | by a shot's end, the audio sits more than 40 ms off the picture |
|
|
861
|
+
| `sync_length` | `analyze_sync_drift` | the soundtrack and the picture differ in length by more than 40 ms |
|
|
862
|
+
|
|
863
|
+
A `shots` record whose `start_frame`/`num_frames` already reaches past the
|
|
864
|
+
file's own length is not measured against a threshold - it is clipped to the
|
|
865
|
+
file before any of the above run, and that clip is itself a `shot_span_overrun`
|
|
866
|
+
finding on all three probes (`analyze_shots`, `analyze_seams`,
|
|
867
|
+
`analyze_sync_drift`), with `value` naming how far past the end it reached.
|
|
868
|
+
`validate_workflow` catches the same mistake before the run for a `shots`
|
|
869
|
+
argument and an `asset:`/literal video whose length is knowable ahead of
|
|
870
|
+
time; it cannot for `previous_result:`/`output:` video not yet written, so
|
|
871
|
+
that case is left to the finding above.
|
|
872
|
+
|
|
873
|
+
**Authority.** A finding marks a place to look, not a verdict. Nothing in
|
|
874
|
+
the engine acts on one, and no run fails because of one. A finding you have
|
|
875
|
+
checked and accepted is simply left alone. A `seam_frame_jump` at a cut the
|
|
876
|
+
story wanted is the cut working, and a level step into a quieter scene can
|
|
877
|
+
be the scene. Tell the person what you confirmed, not what the probe
|
|
878
|
+
reported.
|
|
879
|
+
|
|
880
|
+
**Remediation.** A *recut* reruns only the join over the shots the run
|
|
881
|
+
already made: each entry in `videos` is `output:` + the run's
|
|
882
|
+
`intermediate/` shot file. That is cheap, and generates nothing new. A
|
|
883
|
+
*regenerate* is a new run, so quote its `plan.estimate` first.
|
|
884
|
+
|
|
885
|
+
| Finding | Fix | Kind |
|
|
886
|
+
| --- | --- | --- |
|
|
887
|
+
| `shot_level_spread`, `seam_level_step` | `match_levels: "rms"` (with `match_levels_dbfs` for the target) on the `concat_videos` / `dissolve_videos` step | recut |
|
|
888
|
+
| `seam_click` | a longer `crossfade_ms` on the join | recut |
|
|
889
|
+
| `seam_hole` | `audio_bleed_ms` on the join, so the outgoing tail rings on across the seam | recut |
|
|
890
|
+
| `seam_frame_jump` | a `dissolve_videos` join, or regenerate the incoming shot from the outgoing shot's last frame. If the cut was meant, leave it alone | recut, or regenerate |
|
|
891
|
+
| `sync_drift` | regenerate the shot. Drift inside a shot is the model's, not the join's | regenerate |
|
|
892
|
+
| `sync_length` | rerun the mux through `pair_audio` with `fit: "video"`, which cuts or pads the track to the picture | recut |
|
|
893
|
+
|
|
894
|
+
## Result Configuration
|
|
895
|
+
|
|
896
|
+
```json
|
|
897
|
+
"result": {
|
|
898
|
+
"content_type": "image/jpeg",
|
|
899
|
+
"save": true,
|
|
900
|
+
"file_base_name": "episode",
|
|
901
|
+
"subfolder": "final"
|
|
902
|
+
}
|
|
903
|
+
```
|
|
904
|
+
|
|
905
|
+
Supported content types: `image/jpeg`, `image/png`, `image/webp`, `image/gif`, `video/mp4`, `audio/wav`, `audio/flac`, `audio/mpeg` (mp3), `audio/ogg`, `audio/opus`, `audio/aiff`, `application/json`, `text/plain` (plus the common aliases `audio/x-wav`, `audio/mp3`, `audio/vorbis`).
|
|
906
|
+
|
|
907
|
+
A task command's implementation declares what it hands back - most answer an
|
|
908
|
+
`artifact` (a file `result` saves in one of the media content types above),
|
|
909
|
+
some (`judge`) answer a bare `scalar` that cannot be saved at all, and some
|
|
910
|
+
(the assessment probes in [TASKS.md](TASKS.md)) answer a `json` document -
|
|
911
|
+
every measurement taken, in one dict. A step on a `json` command must set
|
|
912
|
+
`content_type` to `application/json`, which saves it as one document; a step on a `scalar` command
|
|
913
|
+
may not carry a `result` at all. Both are checked in validation, by the
|
|
914
|
+
command's own declared kind rather than a name match.
|
|
915
|
+
|
|
916
|
+
`subfolder` places the step's files in a subfolder of the run directory - see *Saying which output is the deliverable* above. `file_base_name` is the base name the step's files are written under, replacing the name derived from the workflow and step; it may not contain a path separator.
|
|
917
|
+
|
|
918
|
+
For video, `"fps"` is the rate the file is written at. It is rarely needed:
|
|
919
|
+
frames that know their own rate carry it - a video read from a file or an
|
|
920
|
+
`asset:`, a `concat_videos`/`dissolve_videos` join, an interpolation - and
|
|
921
|
+
the engine writes them at it. Frames that bring no rate (most generations)
|
|
922
|
+
fall back to 8, so a workflow that assembles from bare frames should say
|
|
923
|
+
what they run at. A declared `fps` always wins over the carried one and
|
|
924
|
+
warns when the two differ, which is how a deliberate slow motion is written.
|
|
925
|
+
For audio, add `"sample_rate": 44100` when the waveform doesn't
|
|
926
|
+
already carry a rate of its own (a declared rate always wins). Setting `embed_metadata: true`
|
|
927
|
+
on an image result embeds the step's model name and arguments as generation metadata -
|
|
928
|
+
PNG info chunks for `image/png`, EXIF `UserComment` (via `piexif`) for `image/jpeg` and
|
|
929
|
+
`image/webp`.
|
|
930
|
+
|
|
931
|
+
A pipeline that generates a video with its own audio track (LTX-2, or a modular pipeline
|
|
932
|
+
whose `output` asks for both `videos` and `audio`) is muxed into one `video/mp4` file
|
|
933
|
+
with PyAV. `audio_sample_rate` overrides the rate the pipeline itself reports, for the
|
|
934
|
+
rare case it needs correcting.
|
|
935
|
+
|
|
936
|
+
### Audio Encoding
|
|
937
|
+
|
|
938
|
+
Audio is written through soundfile, so both lossless and compressed containers work:
|
|
939
|
+
|
|
940
|
+
```json
|
|
941
|
+
"result": {
|
|
942
|
+
"content_type": "audio/mpeg",
|
|
943
|
+
"sample_rate": 44100,
|
|
944
|
+
"compression_level": 0.3
|
|
945
|
+
}
|
|
946
|
+
```
|
|
947
|
+
|
|
948
|
+
- `subtype` — encoding subtype, such as `"PCM_24"` for wav and flac. Defaults to the
|
|
949
|
+
container's own default, which is `"PCM_16"` for wav and flac.
|
|
950
|
+
- `compression_level` — 0.0 to 1.0 for flac, mp3 and ogg. Higher means smaller files.
|
|
951
|
+
- `bitrate_mode` — `"CONSTANT"`, `"AVERAGE"` or `"VARIABLE"` for compressed formats.
|
|
952
|
+
|
|
953
|
+
`audio/opus` writes an Opus stream in an ogg container, and only encodes at sample rates
|
|
954
|
+
of 8000, 12000, 16000, 24000 or 48000.
|
|
955
|
+
|
|
956
|
+
Output files are saved as `{output_dir}/{base_name}-{result_index}.{artifact_index}.{ext}`,
|
|
957
|
+
where `base_name` is `{workflow_id}-{step_name}.{step_index}` unless the step's result sets
|
|
958
|
+
`file_base_name`, which replaces it entirely. `step_index` is the step's position in the
|
|
959
|
+
workflow, `result_index` counts the argument-combination iterations the step ran (see
|
|
960
|
+
cartesian product, above), and `artifact_index` counts multiple artifacts within one result
|
|
961
|
+
(`num_images_per_prompt > 1`, or a dict result saved key by key). The derived name is what
|
|
962
|
+
makes two steps' files distinct, so when you replace it on more than one step in the same
|
|
963
|
+
subfolder, give each a different name - otherwise the second one gets a `-2` counter.
|
|
964
|
+
|
|
965
|
+
## Pipeline Configuration
|
|
966
|
+
|
|
967
|
+
A step's `configuration` is dw's own vocabulary rather than the model's — each key drives
|
|
968
|
+
a different call — so it is a closed set: a name the schema does not declare fails
|
|
969
|
+
validation instead of being ignored. That matters most for the keys it would otherwise
|
|
970
|
+
be quietest about. A misspelled `offload` used to validate, load, and run with no
|
|
971
|
+
offloading at all, surfacing as an out-of-memory error with nothing pointing at the
|
|
972
|
+
spelling; it now fails before the first model loads. Model-side values that are not part
|
|
973
|
+
of this vocabulary have blocks of their own: `from_pretrained_arguments` for the
|
|
974
|
+
constructor, `arguments` for the call, and `configs` for a modular pipeline's block
|
|
975
|
+
configs.
|
|
976
|
+
|
|
977
|
+
### Memory Offloading
|
|
978
|
+
|
|
979
|
+
Control how models use memory:
|
|
980
|
+
|
|
981
|
+
```json
|
|
982
|
+
"configuration": {
|
|
983
|
+
"component_type": "FluxPipeline",
|
|
984
|
+
"offload": "model"
|
|
985
|
+
}
|
|
986
|
+
```
|
|
987
|
+
|
|
988
|
+
- `"model"` — Moves entire models between CPU and GPU. Good balance of speed and memory.
|
|
989
|
+
- `"sequential"` — Moves individual layers. Slowest but uses least GPU memory. On MPS it is downgraded to `"model"` with a warning: with unified memory there is no separate pool to keep small, so the per-layer copies cost speed and save nothing.
|
|
990
|
+
`exclude_from_cpu_offload` names components the sweep should leave alone.
|
|
991
|
+
- Omit for no offloading (fastest, requires enough VRAM).
|
|
992
|
+
|
|
993
|
+
For components the pipeline loads itself — which is all of a modular pipeline's — use
|
|
994
|
+
`components`, applied once the pipeline is loaded:
|
|
995
|
+
|
|
996
|
+
```json
|
|
997
|
+
"configuration": {
|
|
998
|
+
"component_type": "ModularPipeline",
|
|
999
|
+
"components": {
|
|
1000
|
+
"transformer": {
|
|
1001
|
+
"group_offload": {
|
|
1002
|
+
"offload_type": "block_level",
|
|
1003
|
+
"num_blocks_per_group": 1,
|
|
1004
|
+
"use_stream": true
|
|
1005
|
+
}
|
|
1006
|
+
},
|
|
1007
|
+
"text_encoder.model": {
|
|
1008
|
+
"group_offload": { "offload_type": "leaf_level", "use_stream": true }
|
|
1009
|
+
},
|
|
1010
|
+
"vae": { "device": "cuda", "residency": "on_demand" },
|
|
1011
|
+
"audio_vae": { "device": "cuda" }
|
|
1012
|
+
}
|
|
1013
|
+
}
|
|
1014
|
+
```
|
|
1015
|
+
|
|
1016
|
+
- `group_offload` — streams the component between system memory and the accelerator a
|
|
1017
|
+
block or a leaf module at a time, which is what fits a component larger than the
|
|
1018
|
+
device. `offload_type` is required (`"block_level"` or `"leaf_level"`);
|
|
1019
|
+
`onload_device` defaults to the pipeline's device and `offload_device` to the CPU.
|
|
1020
|
+
Anything else in the block is passed through to `apply_group_offloading`, so
|
|
1021
|
+
`use_stream`, `num_blocks_per_group`, `low_cpu_mem_usage` and
|
|
1022
|
+
`offload_to_disk_path` work as diffusers documents them.
|
|
1023
|
+
- `device` — moves a component that is small enough to stay resident.
|
|
1024
|
+
- `residency` — `"resident"` (the default) leaves the component on its device for the
|
|
1025
|
+
whole run; `"on_demand"` rests it in system memory and moves it to the device only
|
|
1026
|
+
while one of its own calls runs. See [On-demand components](#on-demand-components).
|
|
1027
|
+
- `enable_tiling` — tiled decode for a decoder not named `vae` (LTX-2.5's
|
|
1028
|
+
`diffusion_decoder`, for example).
|
|
1029
|
+
- `attention_backend` — a persistent `set_attention_backend` on one component, which a
|
|
1030
|
+
compiled component needs (the pipeline-level `attention_backend` applies per call).
|
|
1031
|
+
- `attn_processor_type` — the attention processor the component runs, constructed with no
|
|
1032
|
+
arguments and handed to `set_attn_processor`. The `unet` and `transformer` blocks cover
|
|
1033
|
+
those two; this covers any other component that carries attention (LTX-2.5's
|
|
1034
|
+
`diffusion_decoder`, whose default processor is a portable fallback rather than the
|
|
1035
|
+
NATTEN path the decoder was built around).
|
|
1036
|
+
- `compile`, `truncate_layers`, `remove_modules` — see
|
|
1037
|
+
[ACCELERATION.md](ACCELERATION.md).
|
|
1038
|
+
- A dotted key reaches a module inside a component, for a component that holds the model
|
|
1039
|
+
rather than being one.
|
|
1040
|
+
- A `components` block that group offloads anything, or marks anything `on_demand`,
|
|
1041
|
+
already keeps the pipeline itself off the device - the components are placed
|
|
1042
|
+
individually, so moving the whole pipeline would load it in full before the hooks and
|
|
1043
|
+
wrappers exist. Nothing extra is needed for that.
|
|
1044
|
+
|
|
1045
|
+
`preserve_device_placement` covers the case that is left: a component loaded already
|
|
1046
|
+
placed, which must not be moved afterwards. A `device_map` load or a quantization that
|
|
1047
|
+
pins its tensors to one device is the usual reason.
|
|
1048
|
+
|
|
1049
|
+
```json
|
|
1050
|
+
"transformer": {
|
|
1051
|
+
"configuration": {
|
|
1052
|
+
"component_type": "FluxTransformer2DModel",
|
|
1053
|
+
"preserve_device_placement": true
|
|
1054
|
+
},
|
|
1055
|
+
"from_pretrained_arguments": {
|
|
1056
|
+
"model_name": "black-forest-labs/FLUX.1-dev",
|
|
1057
|
+
"subfolder": "transformer",
|
|
1058
|
+
"device_map": "cuda"
|
|
1059
|
+
}
|
|
1060
|
+
}
|
|
1061
|
+
```
|
|
1062
|
+
|
|
1063
|
+
> **Renamed:** this setting was `do_not_send_to_device`. The old name is no longer
|
|
1064
|
+
> recognized - a workflow still using it will load the component and then move it to the
|
|
1065
|
+
> device anyway, since an unknown key is ignored rather than rejected. Rename the key.
|
|
1066
|
+
|
|
1067
|
+
#### On-demand components
|
|
1068
|
+
|
|
1069
|
+
`"residency": "on_demand"` sits between the two placements above. A `device` component
|
|
1070
|
+
holds VRAM for the whole run, wasted on a component used twice; group offloading
|
|
1071
|
+
streams per submodule forward, so it restreams the model once per call of every leaf -
|
|
1072
|
+
ruinous for a VAE, whose tiled decode calls its blocks once per tile. On-demand moves the
|
|
1073
|
+
model as a whole around each call, so a tiling loop sits inside a single pair of
|
|
1074
|
+
transfers.
|
|
1075
|
+
|
|
1076
|
+
```json
|
|
1077
|
+
"components": {
|
|
1078
|
+
"vae": { "device": "cuda", "residency": "on_demand" },
|
|
1079
|
+
"audio_vae": { "device": "cuda", "residency": "on_demand" }
|
|
1080
|
+
}
|
|
1081
|
+
```
|
|
1082
|
+
|
|
1083
|
+
The component rests on the CPU and is moved to `device` around whichever of `forward`,
|
|
1084
|
+
`encode` and `decode` it defines, then moved back and the freed VRAM released to the
|
|
1085
|
+
driver. Nested calls are counted, so a `decode` that calls `forward` internally is moved
|
|
1086
|
+
once, not twice.
|
|
1087
|
+
|
|
1088
|
+
- **Use it for a component that is large but called a handful of times** - a VAE that
|
|
1089
|
+
encodes references at the start and decodes the result at the end. Freeing it for the
|
|
1090
|
+
denoise loop is the whole point.
|
|
1091
|
+
- **Not for a component called every step.** A denoising transformer would pay per-call
|
|
1092
|
+
transfers 20-50 times; group offloading is the tool for those.
|
|
1093
|
+
- **Cannot be combined with `group_offload`** on the same component - a group offloaded
|
|
1094
|
+
module holds one group at a time and ignores whole-model moves, so the two cannot both
|
|
1095
|
+
own its placement. Configuring both is rejected at load.
|
|
1096
|
+
- **Ignored when the component's device is the CPU**, where there is nothing to move it
|
|
1097
|
+
off of.
|
|
1098
|
+
|
|
1099
|
+
On a 24GB card, `templates/minimax/reference-to-video.json` peaks at 18.9GiB of reserved VRAM with on-demand
|
|
1100
|
+
VAEs against 23.2GiB resident, and the tighter resident fit costs 40 allocator retries -
|
|
1101
|
+
cache flushes forced by a failed allocation - where the on-demand run has none. The
|
|
1102
|
+
headroom is also what lets the chained variant run: its later segments carry an extra
|
|
1103
|
+
reference and need ~1.9GiB more than the first.
|
|
1104
|
+
|
|
1105
|
+
The same holds for the frame-conditioned workflows. Generating 124 frames at 960x544
|
|
1106
|
+
from a keyframe, with everything else held equal:
|
|
1107
|
+
|
|
1108
|
+
| VAE placement | peak reserved | allocator retries |
|
|
1109
|
+
| ------------- | ------------- | ----------------- |
|
|
1110
|
+
| resident | 22.71GiB | 22 |
|
|
1111
|
+
| on-demand | 18.03GiB | 0 |
|
|
1112
|
+
|
|
1113
|
+
The resident run also logs a `memory mapping failed with OOM` warning per retry, with as
|
|
1114
|
+
little as 3MB free while it tries to map 20MB. It completes - the allocator flushes its
|
|
1115
|
+
cache and succeeds on the retry - but each one is a synchronising stall, and a run that
|
|
1116
|
+
close to the limit fails outright on any workload that needs slightly more. Every
|
|
1117
|
+
MiniMax H3 example uses on-demand VAEs for this reason.
|
|
1118
|
+
|
|
1119
|
+
**Example:** [reference-to-video.json](../workflows/templates/minimax/reference-to-video.json),
|
|
1120
|
+
[image-to-video.json](../workflows/templates/minimax/image-to-video.json)
|
|
1121
|
+
|
|
1122
|
+
#### Releasing a pipeline mid-workflow
|
|
1123
|
+
|
|
1124
|
+
Pipelines stay loaded for the whole run (and across REPL runs) so repeated steps reuse
|
|
1125
|
+
them. When a workflow chains two large models that cannot both fit - generate with one,
|
|
1126
|
+
upscale with another - release the first once its step completes instead of configuring
|
|
1127
|
+
offload on everything:
|
|
1128
|
+
|
|
1129
|
+
```json
|
|
1130
|
+
{
|
|
1131
|
+
"name": "generate",
|
|
1132
|
+
"release_pipeline": true,
|
|
1133
|
+
"pipeline": { ... }
|
|
1134
|
+
}
|
|
1135
|
+
```
|
|
1136
|
+
|
|
1137
|
+
The step-level `release_pipeline` flag unloads the step's pipeline after its results are
|
|
1138
|
+
saved. A later `pipeline_reference` to a released step is an error, and the REPL's
|
|
1139
|
+
cross-run cache will not retain it.
|
|
1140
|
+
|
|
1141
|
+
#### Releasing task models mid-workflow
|
|
1142
|
+
|
|
1143
|
+
Task models - the checkpoints behind `text_generation`, `segment`, `depth_estimator` and
|
|
1144
|
+
the rest - are cached separately from pipelines, so that a step running its task once per
|
|
1145
|
+
result does not reload the same weights on every iteration. Nothing evicts that cache
|
|
1146
|
+
during a run, which matters when a task loads a large model on the device ahead of a
|
|
1147
|
+
generation step: a prompt-expanding language model would hold its weights for the whole
|
|
1148
|
+
run. `release_models` clears it once the step completes:
|
|
1149
|
+
|
|
1150
|
+
```json
|
|
1151
|
+
{
|
|
1152
|
+
"name": "expand_prompt",
|
|
1153
|
+
"release_models": true,
|
|
1154
|
+
"workflow": { "path": "builtin:h3_context_ir.json", "arguments": { ... } }
|
|
1155
|
+
}
|
|
1156
|
+
```
|
|
1157
|
+
|
|
1158
|
+
The flag applies to any step type, and on a `workflow` step it fires once the whole
|
|
1159
|
+
sub-workflow has finished. It clears every cached task model, not only this step's, and a
|
|
1160
|
+
later step needing one of them reloads it.
|
|
1161
|
+
|
|
1162
|
+
**Example:** [enhance-prompt.json](../workflows/templates/minimax/enhance-prompt.json)
|
|
1163
|
+
|
|
1164
|
+
#### A step nothing reads does not run
|
|
1165
|
+
|
|
1166
|
+
Before the first step executes, the engine drops any step whose result no later step
|
|
1167
|
+
reads and which writes no file, and warns once per dropped step saying which and why.
|
|
1168
|
+
`dialogue-short` cast from portraits that already exist used to run its two Z-Image
|
|
1169
|
+
steps anyway and throw the pictures away - about a minute of GPU per episode on
|
|
1170
|
+
something nothing looked at (#122).
|
|
1171
|
+
|
|
1172
|
+
Four things keep a step:
|
|
1173
|
+
|
|
1174
|
+
- **it saves** - a `result` with a `content_type`, and `save` not `false`. A workflow
|
|
1175
|
+
whose whole point is writing three images references nothing, so this is the rule that
|
|
1176
|
+
keeps elision from being destructive. `"save": false` is how a step says it is
|
|
1177
|
+
scaffolding.
|
|
1178
|
+
- **it is the last step** - it is the run's answer, whatever it declares.
|
|
1179
|
+
- **something reads it** - `previous_result:`/`from_previous_result` (including
|
|
1180
|
+
`previous_result:step.property`), a `gather:` (which is a list of those by the time
|
|
1181
|
+
this runs), a `pipeline_reference` naming it, or a `reused_components` entry naming a
|
|
1182
|
+
component it shares.
|
|
1183
|
+
- Elision is transitive, so dropping a step can drop the step it read in turn.
|
|
1184
|
+
|
|
1185
|
+
`release_pipeline` on an elided step moves onto the last surviving step before it when
|
|
1186
|
+
that step loaded the same pipeline, and `release_models` moves unconditionally - a
|
|
1187
|
+
release that vanished with its step would leak the memory it existed to free. The plan a
|
|
1188
|
+
validate call answers with is computed after elision, so `steps`, `downloads_required`
|
|
1189
|
+
and the cost it quotes are the work that will actually happen, and it lists what was
|
|
1190
|
+
dropped under `elided_steps`; the run manifest records the same list.
|
|
1191
|
+
|
|
1192
|
+
If a step you expected to run is named in the warnings, the usual cause is a reference
|
|
1193
|
+
to it spelled wrong somewhere later, or a step that was meant to declare a `result`.
|
|
1194
|
+
|
|
1195
|
+
### VAE Options
|
|
1196
|
+
|
|
1197
|
+
```json
|
|
1198
|
+
"configuration": {
|
|
1199
|
+
"vae": {
|
|
1200
|
+
"enable_slicing": true,
|
|
1201
|
+
"enable_tiling": true
|
|
1202
|
+
}
|
|
1203
|
+
}
|
|
1204
|
+
```
|
|
1205
|
+
|
|
1206
|
+
- `enable_slicing` — Process VAE in slices to reduce memory
|
|
1207
|
+
- `enable_tiling` — Tile large images through the VAE
|
|
1208
|
+
|
|
1209
|
+
### LoRAs
|
|
1210
|
+
|
|
1211
|
+
Attach one or more LoRAs to a pipeline with `loras`, a sibling of `configuration`:
|
|
1212
|
+
|
|
1213
|
+
```json
|
|
1214
|
+
"loras": [
|
|
1215
|
+
{ "model_name": "XLabs-AI/flux-RealismLora", "adapter_name": "realism", "scale": 0.8 },
|
|
1216
|
+
{ "model_name": "user/other-lora", "weight_name": "lora.safetensors", "subfolder": "loras" }
|
|
1217
|
+
]
|
|
1218
|
+
```
|
|
1219
|
+
|
|
1220
|
+
- `model_name` — the LoRA's hub repo, required.
|
|
1221
|
+
- `weight_name` / `subfolder` — pick a specific weights file within the repo.
|
|
1222
|
+
- `adapter_name` — name passed to `set_adapters()`. Defaults to the LoRA's index in the list.
|
|
1223
|
+
- `scale` — the adapter's weight, passed to `set_adapters()`. Defaults to `1.0`.
|
|
1224
|
+
|
|
1225
|
+
See [workflows/templates/lora.json](../workflows/templates/lora.json) for a full example.
|
|
1226
|
+
|
|
1227
|
+
### IP-Adapter
|
|
1228
|
+
|
|
1229
|
+
```json
|
|
1230
|
+
"ip_adapter": {
|
|
1231
|
+
"model_name": "h94/IP-Adapter",
|
|
1232
|
+
"weight_name": "ip-adapter_sdxl.bin",
|
|
1233
|
+
"scale": 0.6
|
|
1234
|
+
}
|
|
1235
|
+
```
|
|
1236
|
+
|
|
1237
|
+
`model_name` is required; `weight_name`, `subfolder` and `scale` are optional. The
|
|
1238
|
+
adapter image itself is passed as a normal `ip_adapter_image` pipeline argument. See
|
|
1239
|
+
[workflows/templates/ip-adapter.json](../workflows/templates/ip-adapter.json).
|
|
1240
|
+
|
|
1241
|
+
### Sharing Components Across Steps
|
|
1242
|
+
|
|
1243
|
+
Two pipeline steps that load the same underlying component (a shared text encoder, for
|
|
1244
|
+
instance) can avoid loading it twice:
|
|
1245
|
+
|
|
1246
|
+
```json
|
|
1247
|
+
"configuration": { "component_type": "FluxPipeline", "shared_components": ["text_encoder"] }
|
|
1248
|
+
```
|
|
1249
|
+
|
|
1250
|
+
```json
|
|
1251
|
+
"configuration": { "component_type": "FluxPipeline", "reused_components": ["text_encoder"] }
|
|
1252
|
+
```
|
|
1253
|
+
|
|
1254
|
+
The step naming `shared_components` stores those components after it loads; a later step
|
|
1255
|
+
naming the same names in `reused_components` gets them instead of loading its own copy.
|
|
1256
|
+
The names must match exactly between the two steps. Either list can sit in the step's
|
|
1257
|
+
`configuration` or beside it on the pipeline itself.
|
|
1258
|
+
|
|
1259
|
+
How the component reaches the second pipeline depends on what kind it is. A standard
|
|
1260
|
+
pipeline takes it as a `from_pretrained` argument. A modular pipeline cannot — it is
|
|
1261
|
+
built from the component specs in its own index — so it is registered with
|
|
1262
|
+
`update_components()` before `load_components()` runs, which is also what keeps
|
|
1263
|
+
`load_components()` from pulling a second copy: it only loads what is not already there.
|
|
1264
|
+
That is what lets two MiniMax-H3 steps of different tasks (`t2va` and `ref2va` load
|
|
1265
|
+
different transformer partitions) share the 14GB text encoder and the VAEs between them.
|
|
1266
|
+
|
|
1267
|
+
A reused component keeps the device placement the step that shared it gave it. Any
|
|
1268
|
+
`components` entry naming one is skipped with a log line rather than applied a second
|
|
1269
|
+
time — offloading hooks do not survive being installed twice, and the step that loaded
|
|
1270
|
+
the component is the one that decided how it is placed.
|
|
1271
|
+
|
|
1272
|
+
Sharing outlives the pipeline that did it: a step can share a component and still set
|
|
1273
|
+
`release_pipeline`, which frees everything else it loaded while the shared component
|
|
1274
|
+
stays alive for the steps that reuse it.
|
|
1275
|
+
|
|
1276
|
+
### Attention and Performance
|
|
1277
|
+
|
|
1278
|
+
```json
|
|
1279
|
+
"configuration": {
|
|
1280
|
+
"component_type": "FluxPipeline",
|
|
1281
|
+
"attention_backend": "flash_hub",
|
|
1282
|
+
"enable_attention_slicing": true,
|
|
1283
|
+
"no_generator": false
|
|
1284
|
+
}
|
|
1285
|
+
```
|
|
1286
|
+
|
|
1287
|
+
- `enable_attention_slicing` — process attention in slices to reduce memory. Enabled
|
|
1288
|
+
automatically on MPS unless `disable_attention_slicing` is set.
|
|
1289
|
+
- `attention_backend` — selects a diffusers attention backend (e.g. `"flash_hub"`) for
|
|
1290
|
+
the duration of each pipeline call.
|
|
1291
|
+
- `prompt_weighting` — enables A1111-style prompt weighting (`(word:1.5)`, `[word]`,
|
|
1292
|
+
`((word))`) and prompts over 77 tokens. Currently supports Flux pipelines. Mutually
|
|
1293
|
+
exclusive with `remote_text_encoder`.
|
|
1294
|
+
- `no_generator` — set `true` to skip creating a `torch.Generator` for pipelines that
|
|
1295
|
+
don't accept one.
|
|
1296
|
+
- `inversion` — run the pipeline's `invert()` instead of the pipeline itself; the step
|
|
1297
|
+
returns the inverted/image latents for a later step to consume (see
|
|
1298
|
+
[community-pipeline.json](../workflows/templates/community-pipeline.json)).
|
|
1299
|
+
- `generate` — run the pipeline's `generate()` instead, for components with a
|
|
1300
|
+
generation head (the step returns `generated_ids`).
|
|
1301
|
+
|
|
1302
|
+
### Cache Acceleration
|
|
1303
|
+
|
|
1304
|
+
Two mutually exclusive ways to speed up inference by skipping redundant computation:
|
|
1305
|
+
|
|
1306
|
+
```json
|
|
1307
|
+
"configuration": {
|
|
1308
|
+
"cache": { "type": "first_block", "threshold": 0.05 }
|
|
1309
|
+
}
|
|
1310
|
+
```
|
|
1311
|
+
|
|
1312
|
+
`cache` wraps diffusers' own cache hooks - `type` is one of `first_block`, `faster`,
|
|
1313
|
+
`mag`, `taylorseer` or `text_kv`, each with its own tuning fields (`threshold`,
|
|
1314
|
+
`num_inference_steps`, `max_skip_steps`, `retention_ratio`, `cache_interval`,
|
|
1315
|
+
`max_order`, `mag_ratios`, `calibrate` — see [dw/workflow_schema.json](../dw/workflow_schema.json) for which
|
|
1316
|
+
fields apply to which type). See
|
|
1317
|
+
[workflows/templates/step-caching.json](../workflows/templates/step-caching.json).
|
|
1318
|
+
|
|
1319
|
+
```json
|
|
1320
|
+
"configuration": {
|
|
1321
|
+
"teacache": { "rel_l1_thresh": 0.4 }
|
|
1322
|
+
}
|
|
1323
|
+
```
|
|
1324
|
+
|
|
1325
|
+
`teacache` enables TeaCache, currently for Flux transformers, and requires
|
|
1326
|
+
`num_inference_steps` among the pipeline's arguments.
|
|
1327
|
+
|
|
1328
|
+
### Device and Dtype
|
|
1329
|
+
|
|
1330
|
+
Device is auto-detected (CUDA > MPS > CPU). Dtype is set per-component:
|
|
1331
|
+
|
|
1332
|
+
```json
|
|
1333
|
+
"from_pretrained_arguments": {
|
|
1334
|
+
"model_name": "black-forest-labs/FLUX.1-dev",
|
|
1335
|
+
"torch_dtype": "torch.bfloat16"
|
|
1336
|
+
}
|
|
1337
|
+
```
|
|
1338
|
+
|
|
1339
|
+
A step can name a device instead, in a pipeline `configuration` (which becomes the
|
|
1340
|
+
default for that pipeline's components), in a component `configuration`, or in a task's
|
|
1341
|
+
`arguments`. A device naming a backend the machine running the workflow does not have is
|
|
1342
|
+
translated to the one it does, with a warning, so a workflow written on a CUDA box runs
|
|
1343
|
+
on a Mac and back again:
|
|
1344
|
+
|
|
1345
|
+
```json
|
|
1346
|
+
"configuration": {
|
|
1347
|
+
"component_type": "FluxPipeline",
|
|
1348
|
+
"device": "cuda"
|
|
1349
|
+
}
|
|
1350
|
+
```
|
|
1351
|
+
|
|
1352
|
+
Only the backend is translated. A device index survives when the backend matches, so
|
|
1353
|
+
`cuda:1` on a single-GPU CUDA box remains the error it always was; when the backend does
|
|
1354
|
+
not match, the index is dropped and the warning says so — a workflow that meant to spread
|
|
1355
|
+
work across two accelerators will not on a machine that has one. `"device": "cpu"` is
|
|
1356
|
+
never translated, since pinning a step to the CPU is how a GPU-specific problem gets
|
|
1357
|
+
ruled out.
|
|
1358
|
+
|
|
1359
|
+
### Modular Pipelines
|
|
1360
|
+
|
|
1361
|
+
Modular pipelines (`ModularPipeline` and its subclasses) load their configuration and
|
|
1362
|
+
their component weights separately, so `from_pretrained_arguments` only names the model
|
|
1363
|
+
and `load_components` pulls the weights:
|
|
1364
|
+
|
|
1365
|
+
```json
|
|
1366
|
+
"configuration": {
|
|
1367
|
+
"component_type": "MiniMaxMusic3ModularPipeline",
|
|
1368
|
+
"load_components": { "dtype": "torch.bfloat16" },
|
|
1369
|
+
"components_manager": { "enable_auto_cpu_offload": true }
|
|
1370
|
+
}
|
|
1371
|
+
```
|
|
1372
|
+
|
|
1373
|
+
- `load_components` — arguments for `load_components()`. Use `dtype` for the component
|
|
1374
|
+
dtype and `names` to load only some of the components. `quantization_config` is keyed
|
|
1375
|
+
by component name, since a modular pipeline loads each component itself:
|
|
1376
|
+
|
|
1377
|
+
```json
|
|
1378
|
+
"load_components": {
|
|
1379
|
+
"dtype": "torch.bfloat16",
|
|
1380
|
+
"quantization_config": {
|
|
1381
|
+
"transformer": {
|
|
1382
|
+
"configuration": { "config_type": "TorchAoConfig" },
|
|
1383
|
+
"arguments": {
|
|
1384
|
+
"quant_type": "torchao.quantization.Int8WeightOnlyConfig",
|
|
1385
|
+
"modules_to_not_convert": ["proj_in", "proj_out"]
|
|
1386
|
+
}
|
|
1387
|
+
},
|
|
1388
|
+
"language_model": {
|
|
1389
|
+
"configuration": { "config_type": "transformers.TorchAoConfig" },
|
|
1390
|
+
"arguments": { "quant_type": "torchao.quantization.Int8WeightOnlyConfig" }
|
|
1391
|
+
}
|
|
1392
|
+
}
|
|
1393
|
+
}
|
|
1394
|
+
```
|
|
1395
|
+
|
|
1396
|
+
A component the map does not name loads unquantized. Note which `TorchAoConfig` each
|
|
1397
|
+
component takes: the diffusers one for its own models, the transformers one for a
|
|
1398
|
+
transformers model such as a conditioner.
|
|
1399
|
+
- `configs` — values the pipeline's blocks declare and read while they run. They are
|
|
1400
|
+
neither components nor call arguments, which is why they have a block of their own:
|
|
1401
|
+
|
|
1402
|
+
```json
|
|
1403
|
+
"configs": {
|
|
1404
|
+
"canvas_short_edge": 768,
|
|
1405
|
+
"reference_image_short_edge": 1024
|
|
1406
|
+
}
|
|
1407
|
+
```
|
|
1408
|
+
|
|
1409
|
+
The names are whatever the pipeline itself declares, so they differ per model rather
|
|
1410
|
+
than being a fixed list here — MiniMax-H3 declares `canvas_short_edge` (768),
|
|
1411
|
+
`canvas_max_pixels` (1032192) and `reference_image_short_edge` (2048), the last being
|
|
1412
|
+
the resolution its image references are encoded at. A name the pipeline does not
|
|
1413
|
+
declare raises rather than passing quietly, since a dropped config reads as a setting
|
|
1414
|
+
that did nothing.
|
|
1415
|
+
- `components_manager` — attaches a `ComponentsManager`, which tracks the pipeline's
|
|
1416
|
+
components. With `enable_auto_cpu_offload` it keeps only the running components on the
|
|
1417
|
+
device and moves the rest to system memory, reserving `memory_reserve_margin`
|
|
1418
|
+
(default `"3GB"`) of free device memory. It requires a device that reports free memory
|
|
1419
|
+
(CUDA) and replaces `offload`, which modular pipelines do not support.
|
|
1420
|
+
|
|
1421
|
+
A modular pipeline returns whatever its `output` argument asks for — one output by name,
|
|
1422
|
+
or several of them together:
|
|
1423
|
+
|
|
1424
|
+
```json
|
|
1425
|
+
"arguments": {
|
|
1426
|
+
"prompt": "variable:prompt",
|
|
1427
|
+
"output": ["videos", "audio", "sampling_rate"]
|
|
1428
|
+
}
|
|
1429
|
+
```
|
|
1430
|
+
|
|
1431
|
+
Asked for several, the outputs come back keyed by name. Video generated with its own
|
|
1432
|
+
soundtrack is muxed into a single `video/mp4` file, the same way a video pipeline's own
|
|
1433
|
+
output is, and a later step can still reference any of the outputs by name.
|
|
1434
|
+
|
|
1435
|
+
Some repositories hold more than one task's weights. `workflow` names the task, which
|
|
1436
|
+
prunes the pipeline to the blocks that task runs, so only the components it needs are
|
|
1437
|
+
downloaded and loaded:
|
|
1438
|
+
|
|
1439
|
+
```json
|
|
1440
|
+
"from_pretrained_arguments": {
|
|
1441
|
+
"model_name": "MiniMaxAI/MiniMax-H3",
|
|
1442
|
+
"workflow": "t2va"
|
|
1443
|
+
}
|
|
1444
|
+
```
|
|
1445
|
+
|
|
1446
|
+
A task is chosen by the arguments the step passes, so one `workflow` name can cover more
|
|
1447
|
+
than one of them: MiniMax-H3's `fl2va` takes an `image`, a `last_image`, or both. Given
|
|
1448
|
+
only a `last_image` it generates *up to* that frame, inventing everything that leads to
|
|
1449
|
+
it — see [workflows/templates/minimax/last-frame-only.json](../workflows/templates/minimax/last-frame-only.json) beside
|
|
1450
|
+
[workflows/templates/minimax/first-and-last-frame.json](../workflows/templates/minimax/first-and-last-frame.json).
|
|
1451
|
+
|
|
1452
|
+
See [workflows/templates/minimax/music.json](../workflows/templates/minimax/music.json) and
|
|
1453
|
+
[workflows/templates/minimax/video-with-audio.json](../workflows/templates/minimax/video-with-audio.json) for full examples.
|
|
1454
|
+
|
|
1455
|
+
### Chained Video Generation
|
|
1456
|
+
|
|
1457
|
+
Video pipelines generate short clips - a `chain` block on a pipeline step runs the
|
|
1458
|
+
pipeline once per segment and stitches the segments into one long video. The model
|
|
1459
|
+
loads once; each segment's last frame is carried into the next segment as its
|
|
1460
|
+
keyframe, the duplicated boundary frames are trimmed, and frames and audio are
|
|
1461
|
+
joined into a single file:
|
|
1462
|
+
|
|
1463
|
+
```json
|
|
1464
|
+
"pipeline": {
|
|
1465
|
+
"configuration": { "component_type": "LTX2ImageToVideoPipeline" },
|
|
1466
|
+
"from_pretrained_arguments": { "model_name": "Lightricks/LTX-2.5-Diffusers" },
|
|
1467
|
+
"chain": {
|
|
1468
|
+
"segments": 3,
|
|
1469
|
+
"trim_frames": 2,
|
|
1470
|
+
"crossfade_ms": 80
|
|
1471
|
+
},
|
|
1472
|
+
"arguments": { "prompt": "variable:prompt", "image": "variable:image" }
|
|
1473
|
+
}
|
|
1474
|
+
```
|
|
1475
|
+
|
|
1476
|
+
- `segments` — how many times the pipeline runs. Total length is roughly
|
|
1477
|
+
`segments * num_frames`, minus `trim_frames` per seam.
|
|
1478
|
+
- `match_audio` — instead of a count, derive the length from the audio reference in
|
|
1479
|
+
the step's arguments. The audio is sliced into frame-aligned per-segment chunks,
|
|
1480
|
+
each segment is generated against its slice, and the final video is muxed with the
|
|
1481
|
+
**original, unsliced track** - so the soundtrack has no seams at all. Requires
|
|
1482
|
+
`num_frames` (the per-segment length) and a frame rate. Exactly one of `segments`
|
|
1483
|
+
or `match_audio` must be given.
|
|
1484
|
+
- `continuity` — how continuity carries across segments. `last_frame` (the default)
|
|
1485
|
+
extracts each segment's last frame and passes it to the next segment - single-frame
|
|
1486
|
+
conditioning, which carries pose and colour. `last_segment` carries the previous
|
|
1487
|
+
segment itself (frames and its generated soundtrack) into the next as a video
|
|
1488
|
+
reference, which also carries motion, camera, and voice across the seam; it
|
|
1489
|
+
requires a `segment_argument` that takes a references list.
|
|
1490
|
+
- `carry_frames` — with `last_segment`, bound the carry to the last N frames of the
|
|
1491
|
+
segment (the audio is cut to the same span). Unset carries the whole segment.
|
|
1492
|
+
- `carry_audio` — with `last_segment`, whether the carried reference includes its
|
|
1493
|
+
soundtrack (default `true`).
|
|
1494
|
+
- `segment_argument` — where the carried frame or reference lands: `image` (default)
|
|
1495
|
+
for image-to-video pipelines, or `references` for reference-conditioned modular
|
|
1496
|
+
pipelines, where it is appended alongside the workflow's own.
|
|
1497
|
+
- `trim_frames` — image-to-video pipelines reproduce their keyframe as frame 0, so
|
|
1498
|
+
this many frames are dropped from the head of every segment after the first
|
|
1499
|
+
(default 1). The matching audio is used as crossfade material, so video and audio
|
|
1500
|
+
stay exactly in sync. It also bounds the crossfade window: `trim_frames / fps`
|
|
1501
|
+
seconds (at 24 fps, `trim_frames: 2` allows the full default 75 ms fade).
|
|
1502
|
+
- `crossfade_ms` — equal-power crossfade applied to *generated* audio at each seam
|
|
1503
|
+
(default 75). Not used with `match_audio`, which keeps the original track.
|
|
1504
|
+
- `fps` — frame rate for the chain's audio math. Defaults to the pipeline's
|
|
1505
|
+
`frame_rate` argument; pipelines with a fixed rate need it set (MiniMax H3: 24).
|
|
1506
|
+
- `frame_snap` — the constraint the pipeline puts on `num_frames`, used to snap the
|
|
1507
|
+
final `match_audio` segment to a valid length. MiniMax H3 accepts `17n+5` frames
|
|
1508
|
+
between 124 and 345: `{ "modulus": 17, "remainder": 5, "min_frames": 124,
|
|
1509
|
+
"max_frames": 345 }`. Where the workflow already declares that rule as a
|
|
1510
|
+
`variable_constraints` entry, write `"frame_snap": "constraint:num_frames"`
|
|
1511
|
+
instead, so the numbers live in one place (*What a variable is allowed to be*).
|
|
1512
|
+
- `prompts` — optional per-segment prompt list for narrative progression; segment
|
|
1513
|
+
`i` uses `prompts[min(i, len - 1)]`.
|
|
1514
|
+
- `save_segments` — write each completed segment to the output directory as a
|
|
1515
|
+
playable mp4 and free its frames, bounding memory to roughly one segment
|
|
1516
|
+
regardless of chain length. The final video is streamed from the segment files
|
|
1517
|
+
at save time, and they are removed once it is written (`keep_segments: true`
|
|
1518
|
+
retains them). A crashed chain leaves the finished segments behind - stitch
|
|
1519
|
+
them by hand by listing their paths in a `concat_videos` step (`trim_frames: 0`,
|
|
1520
|
+
the trim was already applied). Requires PyAV and a frame rate. The trade-off is
|
|
1521
|
+
one extra encode/decode cycle through h264 for the segment files.
|
|
1522
|
+
|
|
1523
|
+
The chain runs inside one iteration of the step, so it composes with
|
|
1524
|
+
`previous_result` fan-out (three keyframes in, three chained videos out), and a
|
|
1525
|
+
`pipeline_reference` step can carry its own `chain`. Seeds behave like a normal run:
|
|
1526
|
+
the step's generator advances across segments, so one seed reproduces the whole
|
|
1527
|
+
chain. Expect some visual drift across many segments with `last_frame` continuity -
|
|
1528
|
+
it is single-frame conditioning; `last_segment` continuity exists for exactly that,
|
|
1529
|
+
where the pipeline can take a video reference.
|
|
1530
|
+
|
|
1531
|
+
See [workflows/templates/ltx2/chained-segments.json](../workflows/templates/ltx2/chained-segments.json),
|
|
1532
|
+
[workflows/templates/minimax/chained-segments.json](../workflows/templates/minimax/chained-segments.json), and
|
|
1533
|
+
[workflows/templates/minimax/chain-matched-to-audio.json](../workflows/templates/minimax/chain-matched-to-audio.json)
|
|
1534
|
+
(audio-matched lip-sync of arbitrary length).
|
|
1535
|
+
|
|
1536
|
+
## Schedulers
|
|
1537
|
+
|
|
1538
|
+
Override the default scheduler:
|
|
1539
|
+
|
|
1540
|
+
```json
|
|
1541
|
+
"scheduler": {
|
|
1542
|
+
"configuration": {
|
|
1543
|
+
"scheduler_type": "DPMSolverMultistepScheduler"
|
|
1544
|
+
},
|
|
1545
|
+
"from_config_args": {
|
|
1546
|
+
"use_karras_sigmas": true
|
|
1547
|
+
}
|
|
1548
|
+
}
|
|
1549
|
+
```
|
|
1550
|
+
|
|
1551
|
+
A scheduler block may also carry `shift`, the exponential sigma shift for
|
|
1552
|
+
schedulers that take one (MiniMax H3's released checkpoint: 12.0 for video,
|
|
1553
|
+
3.0 for audio). A pipeline that carries a second scheduler takes an
|
|
1554
|
+
`audio_scheduler` block with the same shape - MiniMax H3 steps video and audio
|
|
1555
|
+
latents down two schedules whose shifts are set independently.
|
|
1556
|
+
|
|
1557
|
+
## Seeds
|
|
1558
|
+
|
|
1559
|
+
Set a seed for reproducibility at workflow, step, or pipeline level - most specific wins:
|
|
1560
|
+
a pipeline's own `seed` overrides its step's, which overrides the workflow's:
|
|
1561
|
+
|
|
1562
|
+
```json
|
|
1563
|
+
{
|
|
1564
|
+
"id": "my_workflow",
|
|
1565
|
+
"seed": 42,
|
|
1566
|
+
"steps": [
|
|
1567
|
+
{ "name": "step1", "seed": 123, "pipeline": { "seed": 7, ... } }
|
|
1568
|
+
]
|
|
1569
|
+
}
|
|
1570
|
+
```
|
|
1571
|
+
|
|
1572
|
+
Omit `seed` entirely to let the workflow draw a random one at run time. The seed a run
|
|
1573
|
+
actually used - drawn or named - is recorded in its `manifest.json`, so a run you liked
|
|
1574
|
+
can be reproduced after the fact.
|
|
1575
|
+
|
|
1576
|
+
Beside that manifest the run also writes `workflow.json` — the *realized*
|
|
1577
|
+
workflow, meaning the one that actually ran. Every mutable input is pinned into
|
|
1578
|
+
it: the caller's `arguments` folded into the `variables` defaults, the seed the
|
|
1579
|
+
run used, each `prompt:` reference replaced by the stored text, and each
|
|
1580
|
+
`output:<identity>/latest/<file>` (or `/v<N>/`) rewritten to the run id it resolved to.
|
|
1581
|
+
`asset:`, `constant:`, `previous_result:` and `builtin:` are kept as written —
|
|
1582
|
+
each already names something pinned by the asset library or by the manifest's
|
|
1583
|
+
`dw_version` — and a sub-workflow named by local path is kept with its file's
|
|
1584
|
+
SHA-256 recorded in the manifest. The manifest also lists which stored prompts
|
|
1585
|
+
were inlined, since inlining loses the name.
|
|
1586
|
+
|
|
1587
|
+
A step that joins shots (`concat_videos`, `dissolve_videos`, or a pipeline
|
|
1588
|
+
step with a `chain`) also records where each one landed, as `shots` on its
|
|
1589
|
+
manifest entry (and on its `step_end` event):
|
|
1590
|
+
|
|
1591
|
+
```json
|
|
1592
|
+
{
|
|
1593
|
+
"step": "cut",
|
|
1594
|
+
"files": ["final/film.mp4"],
|
|
1595
|
+
"subfolder": "final",
|
|
1596
|
+
"shots": [
|
|
1597
|
+
{"name": "shot@a", "start_frame": 0, "num_frames": 121, "start_sample": 0, "num_samples": 242267},
|
|
1598
|
+
{"name": "shot@b", "start_frame": 121, "num_frames": 97, "start_sample": 242267, "num_samples": 194000}
|
|
1599
|
+
]
|
|
1600
|
+
}
|
|
1601
|
+
```
|
|
1602
|
+
|
|
1603
|
+
The shots partition the file's frames: the `num_frames` add up to the frame
|
|
1604
|
+
count. The sample fields are *measured* off the track the join built, not
|
|
1605
|
+
worked out from the frames. That means a shot whose track ran long shows it
|
|
1606
|
+
here: the first shot above is 267 samples longer than 121 frames at 24 fps.
|
|
1607
|
+
They are null when the video has no track, and for a chain that uses
|
|
1608
|
+
`match_audio`. A shot is named `shot@<key>` when the step's `videos` entry was
|
|
1609
|
+
a `previous_result:shot@<key>` reference, else by its input's position
|
|
1610
|
+
(`video N`, a chain's `segment N`). A dissolve's shots after the first carry
|
|
1611
|
+
`overlap_frames`, the head they share with the shot before. A step that wrote
|
|
1612
|
+
several joined files marks each shot with its `file`.
|
|
1613
|
+
|
|
1614
|
+
The steps that keep the frames pass `shots` on. `stabilize` and the per-frame
|
|
1615
|
+
tasks keep them as they are. `interpolate_frames` rescales them to the new
|
|
1616
|
+
frame count and clears the samples. `pair_audio` measures the samples again
|
|
1617
|
+
against the new track - every shot but the last is `round(start_frame / fps *
|
|
1618
|
+
sample_rate)`, and the last one runs to the track's actual end - measured
|
|
1619
|
+
again, once the file is written, against what it decodes to. So its
|
|
1620
|
+
`num_samples` can be a few dozen samples off `round(num_frames * sample_rate
|
|
1621
|
+
/ fps)`: the AAC encode's trim, recorded in the job's event log
|
|
1622
|
+
(not a dropped sample - a real
|
|
1623
|
+
mismatch between the track and the video's length is its own warning,
|
|
1624
|
+
`audio_video_length_mismatch` or `audio_padded_to_video/audio_trimmed_to_video`
|
|
1625
|
+
with `fit: "video"`). Everything else drops them: an audio task's track,
|
|
1626
|
+
say, or a video read back from a file. `get_gallery_metadata` reports the
|
|
1627
|
+
recorded shots as `media.shots`, and `get_output_frames(seams=true)` uses them
|
|
1628
|
+
when you pass no `boundaries`.
|
|
1629
|
+
|
|
1630
|
+
The file is a valid workflow, and running it again is `python -m dw.run
|
|
1631
|
+
workflow.json` or handing its contents to `run_workflow` as `inline_workflow`
|
|
1632
|
+
— but either way the `asset:` and `output:` names in it resolve against the
|
|
1633
|
+
server's or CLI's own libraries, not against the run directory, so doing this
|
|
1634
|
+
from inside that directory reproduces the run only when its libraries are the
|
|
1635
|
+
ones the original run used too. Writing the file is best effort, exactly like
|
|
1636
|
+
the manifest — a run that produced its files has succeeded either way — and
|
|
1637
|
+
`--output-layout flat` writes no run directory, so it writes neither file.
|
|
1638
|
+
|
|
1639
|
+
Any of the three levels accepts a `variable:` reference, which is how a seed becomes
|
|
1640
|
+
settable per run without editing the file:
|
|
1641
|
+
|
|
1642
|
+
```json
|
|
1643
|
+
{
|
|
1644
|
+
"variables": { "seed": 42 },
|
|
1645
|
+
"seed": "variable:seed",
|
|
1646
|
+
"steps": [ ... ]
|
|
1647
|
+
}
|
|
1648
|
+
```
|
|
1649
|
+
|
|
1650
|
+
```bash
|
|
1651
|
+
python -m dw.run workflows/models/z-image.json seed=1234
|
|
1652
|
+
```
|
|
1653
|
+
|
|
1654
|
+
Declare the variable with an integer default, as above: the value from the command line
|
|
1655
|
+
arrives as a string and is converted to the declared type. A string that is not a
|
|
1656
|
+
`variable:` reference is rejected by the schema.
|
|
1657
|
+
|
|
1658
|
+
The seed also reaches sub-workflows: a delegated `workflow` step runs the child under
|
|
1659
|
+
the parent's seed unless the child names its own. Without that a child draws its own
|
|
1660
|
+
random seed, and a workflow whose real generation happens inside a sub-workflow would
|
|
1661
|
+
not reproduce from the seed it was given.
|
|
1662
|
+
|
|
1663
|
+
The step cache that lets a reproduced step skip re-running (see *Runs* in
|
|
1664
|
+
[Workspaces](WORKSPACES.md#runs)) is scoped to the output directory a run
|
|
1665
|
+
writes into, which on `dw.serve` is the pinned workspace's own `outputs/`.
|
|
1666
|
+
Two workspaces holding what looks like the same prior run - same workflow,
|
|
1667
|
+
same seed, same arguments - do not share a cache entry, so
|
|
1668
|
+
`validate_workflow`'s `plan.cached_steps` answers for the workspace the call
|
|
1669
|
+
is pinned to, not for every workspace that happens to hold a matching run.
|
|
1670
|
+
Deleting a workspace takes its cache entries with it, the same as deleting
|
|
1671
|
+
its `outputs/` directory would.
|
|
1672
|
+
|
|
1673
|
+
## Type System
|
|
1674
|
+
|
|
1675
|
+
Dynamic type conversion applies to certain values:
|
|
1676
|
+
|
|
1677
|
+
- Keys ending in `_type` or `_dtype`, or named `dtype`: `"torch.bfloat16"` becomes `torch.bfloat16`
|
|
1678
|
+
- Dotted names: `"sdnq.SDNQConfig"` loads the class via importlib
|
|
1679
|
+
- Escape with braces to keep as string: `"{nf4}"` stays as `"nf4"`
|
|
1680
|
+
- `content_type` and `offload_type` are exempt even though they end in `_type` - they
|
|
1681
|
+
name a category, not a Python type, so their value always stays a plain string (the
|
|
1682
|
+
`{}` escape is accepted but not required for these two keys)
|
|
1683
|
+
- Values prefixed with `constant:` are read from python rather than copied into the
|
|
1684
|
+
workflow: `"constant:diffusers.pipelines.ltx2.utils.DISTILLED_SIGMA_VALUES"`
|
|
1685
|
+
|
|
1686
|
+
### Constant References
|
|
1687
|
+
|
|
1688
|
+
Some arguments have a value the library already declares: the sigma schedule a distilled
|
|
1689
|
+
model was trained on, the negative prompt a model family ships. Reference it with
|
|
1690
|
+
`constant:` and its dotted python name instead of copying it into the workflow:
|
|
1691
|
+
|
|
1692
|
+
```json
|
|
1693
|
+
"sigmas": "constant:diffusers.pipelines.ltx2.utils.DISTILLED_SIGMA_VALUES",
|
|
1694
|
+
"negative_prompt": "constant:diffusers.pipelines.ltx2.utils.DEFAULT_NEGATIVE_PROMPT"
|
|
1695
|
+
```
|
|
1696
|
+
|
|
1697
|
+
The leading part of the name that imports is the module, and the rest is read from it -
|
|
1698
|
+
so a constant held in a config object is reachable too:
|
|
1699
|
+
|
|
1700
|
+
```json
|
|
1701
|
+
"prompt_max_new_tokens": "constant:diffusers.pipelines.ltx2.utils.GEMMA4_PROMPT_ENHANCEMENT_CONFIG.max_new_tokens"
|
|
1702
|
+
```
|
|
1703
|
+
|
|
1704
|
+
A reference resolves anywhere in a workflow's arguments, including in a `variables`
|
|
1705
|
+
default, where it becomes the value a caller overrides - and its type, since a variable
|
|
1706
|
+
is declared by its default:
|
|
1707
|
+
|
|
1708
|
+
```json
|
|
1709
|
+
"variables": { "negative_prompt": "constant:diffusers.pipelines.ltx2.utils.DEFAULT_NEGATIVE_PROMPT" }
|
|
1710
|
+
```
|
|
1711
|
+
|
|
1712
|
+
A constant is data. The name has to resolve to a value - anything callable is refused,
|
|
1713
|
+
because a type is named with a `*_type` argument and constructed there, and reaching a
|
|
1714
|
+
function this way would be evaluating python rather than referencing it. Mutable values
|
|
1715
|
+
are copied, so a pipeline that consumes its schedule in place cannot edit the library's
|
|
1716
|
+
constant for the rest of the session.
|
|
1717
|
+
|
|
1718
|
+
The value the library declares is the value the workflow gets, which is the point: a
|
|
1719
|
+
constant that changes upstream changes here, and one that is renamed or moved fails
|
|
1720
|
+
loudly rather than leaving a stale copy behind.
|
|
1721
|
+
|
|
1722
|
+
### Prompt References
|
|
1723
|
+
|
|
1724
|
+
A prompt worth keeping is worth keeping once. Stored prompts live as JSON files in a
|
|
1725
|
+
prompt library - the `prompts/` folder by default - and a workflow argument written as
|
|
1726
|
+
`prompt:` plus the file's name (without `.json`, optionally one folder deep) loads its
|
|
1727
|
+
text at run time:
|
|
1728
|
+
|
|
1729
|
+
```json
|
|
1730
|
+
"prompt": "prompt:scenic_landscape",
|
|
1731
|
+
"prompt": "prompt:minimax/fox_dawn_t2va"
|
|
1732
|
+
```
|
|
1733
|
+
|
|
1734
|
+
A prompt file holds the text plus the metadata the server's Prompts page shows:
|
|
1735
|
+
|
|
1736
|
+
```json
|
|
1737
|
+
{
|
|
1738
|
+
"text": "A sweeping alpine valley at golden hour...",
|
|
1739
|
+
"description": "General-purpose scenic landscape",
|
|
1740
|
+
"intended_model": "z-image",
|
|
1741
|
+
"negative_prompt": "blurry, low quality",
|
|
1742
|
+
"tags": ["landscape", "golden-hour"]
|
|
1743
|
+
}
|
|
1744
|
+
```
|
|
1745
|
+
|
|
1746
|
+
Only `text` is required, and it is what the reference resolves to. `intended_model` is
|
|
1747
|
+
informational - the engine ignores it, but the library badges and filters by it, and
|
|
1748
|
+
the server's prompt enhancer uses it to preselect a preset. One spelling per family:
|
|
1749
|
+
`list_prompts(intended_model=...)` matches the whole value exactly, so `minimax-music`
|
|
1750
|
+
beside `minimax-music3` hides half a shelf, and `tests/test_prompt_library.py` sweeps
|
|
1751
|
+
the repo's library for a variant.
|
|
1752
|
+
|
|
1753
|
+
The library's location is resolved in order: the `DW_PROMPT_DIR` environment
|
|
1754
|
+
variable (which `--prompt-dir` on both `dw.run` and `dw.serve` sets), then
|
|
1755
|
+
`./prompts` in the working directory when it exists, then the first `prompts/`
|
|
1756
|
+
folder found walking up from the workflow file's own directory - which is how
|
|
1757
|
+
a repo workflow run from any working directory still reaches the library beside
|
|
1758
|
+
it. `dw.serve` resolves the directory once at startup with this same order
|
|
1759
|
+
(anchored at its workflow directory) and pins it for every job, so the Prompts
|
|
1760
|
+
page and `prompt:` resolution always agree on one library.
|
|
1761
|
+
References are rooted at that one directory - not at the workflow file - so
|
|
1762
|
+
the same reference means the same text from every workflow. Like `constant:`, a reference
|
|
1763
|
+
resolves anywhere in a workflow's arguments, including a `variables` default, and it
|
|
1764
|
+
always resolves to exactly one string: it never multiplies a step's iterations the way
|
|
1765
|
+
`previous_result:` references do. A prompt's text may not itself begin with a
|
|
1766
|
+
reference prefix such as `variable:` - the engine refuses it rather than resolving
|
|
1767
|
+
text as syntax.
|
|
1768
|
+
|
|
1769
|
+
### Asset References
|
|
1770
|
+
|
|
1771
|
+
A workflow's plain media paths resolve against the workflow file's own directory, which
|
|
1772
|
+
means a workflow that reads anything has to keep that thing beside it. An `asset:`
|
|
1773
|
+
reference is rooted at the asset library instead - the workspace's `assets/` folder -
|
|
1774
|
+
so a workflow and the media it reads do not have to live in the same place:
|
|
1775
|
+
|
|
1776
|
+
```json
|
|
1777
|
+
"image": "asset:iris.png",
|
|
1778
|
+
"video": "asset:gyre/frames/web.mp4",
|
|
1779
|
+
"references": [
|
|
1780
|
+
{
|
|
1781
|
+
"reference_type": "diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference",
|
|
1782
|
+
"from_file": "asset:subject.png"
|
|
1783
|
+
}
|
|
1784
|
+
]
|
|
1785
|
+
```
|
|
1786
|
+
|
|
1787
|
+
A reference names a file with its extension, at most four folders deep, and resolves to
|
|
1788
|
+
that file's path - so it works under any argument that accepts a path: `image`, `video`,
|
|
1789
|
+
a `from_file`, a list of any of them, or a task argument that names a file. What loads
|
|
1790
|
+
the path is unchanged; only where the path comes from is.
|
|
1791
|
+
|
|
1792
|
+
The library's location is resolved in order: the `DW_ASSET_DIR` environment variable
|
|
1793
|
+
(which `--asset-dir` on both `dw.run` and `dw.serve` sets), then the workspace's
|
|
1794
|
+
`assets/` when a workspace was named explicitly, then `./assets` in the working
|
|
1795
|
+
directory when it exists, then the first `assets/` folder found walking up from the
|
|
1796
|
+
workflow file's own directory. See [Workspaces](WORKSPACES.md).
|
|
1797
|
+
|
|
1798
|
+
A reference can only name a file inside the library: `..`, an absolute path, or a
|
|
1799
|
+
symlink pointing out of it are all refused. Browser uploads land in the library's
|
|
1800
|
+
`uploads/` folder and come back as `asset:uploads/<name>`, so a workflow saved after
|
|
1801
|
+
an upload still resolves on the next run.
|
|
1802
|
+
|
|
1803
|
+
### Output References
|
|
1804
|
+
|
|
1805
|
+
Multi-stage work — generate stills, then animate them; generate a score, then mux it —
|
|
1806
|
+
used to mean copying files out of the output directory and back in beside the next
|
|
1807
|
+
workflow. An `output:` reference names what an earlier run wrote, directly:
|
|
1808
|
+
|
|
1809
|
+
```json
|
|
1810
|
+
"image": "output:ltx2/Gyre/latest/Gyre-still.0-0.0.png",
|
|
1811
|
+
"audio": "output:ltx2/GyreScore/20260905-181530-a1b2c3d4/Gyre-score.10-0.0.wav"
|
|
1812
|
+
```
|
|
1813
|
+
|
|
1814
|
+
The name is a path under the output directory — the workflow's identity, the run, and
|
|
1815
|
+
the file (see [Runs](WORKSPACES.md#runs)). Writing `latest` where the run id goes
|
|
1816
|
+
resolves to the newest run of that workflow *that holds the file*, which is what lets a
|
|
1817
|
+
second-stage workflow name the first stage's product without being edited after every
|
|
1818
|
+
run - and keeps working when the newest run failed part way, or reused every step from
|
|
1819
|
+
the cache and so wrote nothing of its own but a manifest. Runs sort by their id, which
|
|
1820
|
+
starts with a UTC timestamp, so "newest" needs no file timestamps and survives a
|
|
1821
|
+
directory being copied. `v<N>` in the same position names the run whose version is N -
|
|
1822
|
+
the `v4` the gallery labels its files with - so the number a person was told is a name
|
|
1823
|
+
a workflow can take. Unlike `latest` it picks exactly one run: `v4` not holding the file
|
|
1824
|
+
is an error, not a reason to try `v3`. `latest` and `v<N>` only select a run where run
|
|
1825
|
+
directories are; a workflow or file that happens to be called either is still named as
|
|
1826
|
+
itself.
|
|
1827
|
+
|
|
1828
|
+
Like `asset:`, a reference resolves to a path and then whatever loads paths loads it, so
|
|
1829
|
+
it works under `image`, `video`, a `from_file`, or a list of them. The audio tasks take
|
|
1830
|
+
a video file's path too and use the soundtrack muxed into it, which is how a finished
|
|
1831
|
+
cut is scored in a later run without re-cutting it. It resolves against
|
|
1832
|
+
the output directory the run was told to write to, and cannot leave it: `..`, an
|
|
1833
|
+
absolute path, and a symlink pointing out are all refused.
|
|
1834
|
+
|
|
1835
|
+
To name an *earlier step of the same run*, use `previous_result:` instead — that passes
|
|
1836
|
+
the value in memory rather than through the filesystem.
|
|
1837
|
+
|
|
1838
|
+
A generated file worth reusing repeatedly is better *kept* than referenced by the run
|
|
1839
|
+
that made it: `POST /api/assets/keep` (the gallery's **Keep as asset**, or MCP's
|
|
1840
|
+
`keep_output`) copies it into the workspace's asset library under a name you choose, and
|
|
1841
|
+
from then on it is an `asset:` reference like any other — stable whatever happens to the
|
|
1842
|
+
run directory it came from.
|
|
1843
|
+
|
|
1844
|
+
### Objects Built From a File
|
|
1845
|
+
|
|
1846
|
+
Some pipelines take arguments that are objects rather than plain media. An argument that
|
|
1847
|
+
names a type and a `from_file` is constructed by that type's own `from_file()`:
|
|
1848
|
+
|
|
1849
|
+
```json
|
|
1850
|
+
"references": [
|
|
1851
|
+
{
|
|
1852
|
+
"reference_type": "diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference",
|
|
1853
|
+
"from_file": "subject.png"
|
|
1854
|
+
},
|
|
1855
|
+
{
|
|
1856
|
+
"reference_type": "diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference",
|
|
1857
|
+
"from_file": "voice.wav"
|
|
1858
|
+
}
|
|
1859
|
+
]
|
|
1860
|
+
```
|
|
1861
|
+
|
|
1862
|
+
Loading the media this way rather than as a plain `image` or `video` argument is what
|
|
1863
|
+
brings its frame rate or sample rate along with it, which MiniMax-H3 resamples a
|
|
1864
|
+
reference from. The file may be a path — relative to the workflow file, like all media a
|
|
1865
|
+
workflow names — or a URL, and is validated like any other media. `variable:` references
|
|
1866
|
+
work as the file location; `previous_result:` does not, since the object is built when
|
|
1867
|
+
the workflow loads — use
|
|
1868
|
+
[`from_previous_result`](#objects-built-from-an-earlier-step) for that. A dict that
|
|
1869
|
+
merely contains a `from_file` key without a `*_type` key is not an object description
|
|
1870
|
+
and is passed through untouched.
|
|
1871
|
+
|
|
1872
|
+
An entry in a list whose source is `null` is **left out** of that list. That is what
|
|
1873
|
+
makes a reference optional: write it as an ordinary entry whose `from_file` (or
|
|
1874
|
+
`from_previous_result`) is a variable, declare the variable `null`, and a run that is
|
|
1875
|
+
given nothing for it generates exactly as it did before the reference existed — one
|
|
1876
|
+
workflow serving both, instead of two spellings of the same steps. It applies to
|
|
1877
|
+
`from_file`, `from_previous_result` and `from_arguments` alike. On its own rather than
|
|
1878
|
+
in a list there is nothing to leave it out of, so a null source there is an error.
|
|
1879
|
+
|
|
1880
|
+
Any other key goes wherever the type can take it: to `from_file()` where its signature
|
|
1881
|
+
names it, and onto the object it returns where it does not. That is what corrects a
|
|
1882
|
+
decoded file, which is the only thing that knows what the container claimed:
|
|
1883
|
+
|
|
1884
|
+
```json
|
|
1885
|
+
{
|
|
1886
|
+
"reference_type": "diffusers.modular_pipelines.minimax_h3.MiniMaxH3VideoReference",
|
|
1887
|
+
"from_file": "motion.mp4",
|
|
1888
|
+
"fps": 30.0,
|
|
1889
|
+
"audio": null
|
|
1890
|
+
}
|
|
1891
|
+
```
|
|
1892
|
+
|
|
1893
|
+
`fps` overrides a rate the container got wrong — MiniMax-H3 resamples a reference onto
|
|
1894
|
+
its own 24 fps, so a wrong rate is a request conditioned at the wrong speed — and
|
|
1895
|
+
`audio: null` drops the decoded soundtrack, leaving a reference that conditions on
|
|
1896
|
+
motion and camera alone. A name that is neither an argument of `from_file()` nor a field
|
|
1897
|
+
of the object raises, with the fields it does have.
|
|
1898
|
+
|
|
1899
|
+
See [workflows/templates/minimax/reference-to-video.json](../workflows/templates/minimax/reference-to-video.json) for a full example.
|
|
1900
|
+
|
|
1901
|
+
### Objects Built From an Earlier Step
|
|
1902
|
+
|
|
1903
|
+
The same object can be built from what an earlier step generated, by naming the step
|
|
1904
|
+
instead of a file:
|
|
1905
|
+
|
|
1906
|
+
```json
|
|
1907
|
+
"references": [
|
|
1908
|
+
{
|
|
1909
|
+
"reference_type": "diffusers.modular_pipelines.minimax_h3.MiniMaxH3ImageReference",
|
|
1910
|
+
"from_previous_result": "draw_subject"
|
|
1911
|
+
}
|
|
1912
|
+
]
|
|
1913
|
+
```
|
|
1914
|
+
|
|
1915
|
+
`from_file` cannot do this — it names a file, and the object is built when the workflow
|
|
1916
|
+
loads, before any step has run. `from_previous_result` waits: the description is checked
|
|
1917
|
+
at load time and constructed once the step it names has produced its media, which is
|
|
1918
|
+
what lets one workflow generate a subject and then condition on it without writing it
|
|
1919
|
+
out and reading it back.
|
|
1920
|
+
|
|
1921
|
+
The media never touches the disk, so it arrives as the step produced it. Which field it
|
|
1922
|
+
lands in comes from the type's own `kind`:
|
|
1923
|
+
|
|
1924
|
+
| `kind` | Built from |
|
|
1925
|
+
| ------- | ------------------------------------------------------------------------- |
|
|
1926
|
+
| `image` | The generated image |
|
|
1927
|
+
| `video` | The generated frames, and the soundtrack generated with them if there was one |
|
|
1928
|
+
| `audio` | The generated soundtrack - or, for a step that produced audio alone (a music pipeline, a `slice_audio` task), the waveform itself. The rate travels with the waveform when the pipeline or task reports one (an `AudioTrack` - AudioLDM2, StableAudio, `generate_speech`); declare `sample_rate` beside `from_previous_result` only for a waveform from a task or file that carries none, and a declared rate always wins |
|
|
1929
|
+
|
|
1930
|
+
Any other key is a field of the object and wins over what the media carried —
|
|
1931
|
+
`"fps": 30.0` where the producing pipeline generated at a rate the consuming one does
|
|
1932
|
+
not share, for instance. A step that produced several artifacts fans out the same way
|
|
1933
|
+
every `previous_result` reference does: four images in, four videos out.
|
|
1934
|
+
|
|
1935
|
+
See [workflows/templates/minimax/generated-subject-reference.json](../workflows/templates/minimax/generated-subject-reference.json)
|
|
1936
|
+
for a full example.
|
|
1937
|
+
|
|
1938
|
+
### Objects Built From Named Arguments
|
|
1939
|
+
|
|
1940
|
+
Not every type a pipeline takes knows how to open a file. LTX-2's keyframe conditions
|
|
1941
|
+
and IC-LoRA references are plain dataclasses holding frames the caller already loaded,
|
|
1942
|
+
plus the numbers that say what to do with them. Those are written as the arguments to
|
|
1943
|
+
construct the object with:
|
|
1944
|
+
|
|
1945
|
+
```json
|
|
1946
|
+
"conditions": [
|
|
1947
|
+
{
|
|
1948
|
+
"condition_type": "diffusers.pipelines.ltx2.pipeline_ltx2_condition.LTX2VideoCondition",
|
|
1949
|
+
"from_arguments": {
|
|
1950
|
+
"frames": { "media_type": "image", "location": "first.png" },
|
|
1951
|
+
"index": 0,
|
|
1952
|
+
"strength": 1.0
|
|
1953
|
+
}
|
|
1954
|
+
},
|
|
1955
|
+
{
|
|
1956
|
+
"condition_type": "diffusers.pipelines.ltx2.pipeline_ltx2_condition.LTX2VideoCondition",
|
|
1957
|
+
"from_arguments": {
|
|
1958
|
+
"frames": { "media_type": "image", "location": "last.png" },
|
|
1959
|
+
"index": -1,
|
|
1960
|
+
"strength": 1.0
|
|
1961
|
+
}
|
|
1962
|
+
}
|
|
1963
|
+
]
|
|
1964
|
+
```
|
|
1965
|
+
|
|
1966
|
+
`from_arguments` holds every argument the type is constructed with - a key beside it
|
|
1967
|
+
raises rather than being silently dropped, and so does an argument the type does not
|
|
1968
|
+
take, naming the ones it does. The arguments inside are ordinary arguments: a
|
|
1969
|
+
[media reference](#media-arguments) loads there, a `variable:` reference resolves
|
|
1970
|
+
there, and a `previous_result:` reference waits the way
|
|
1971
|
+
[`from_previous_result`](#objects-built-from-an-earlier-step) does - the object is
|
|
1972
|
+
constructed once the step it names has run.
|
|
1973
|
+
|
|
1974
|
+
Which of the three forms a type wants is decided by the type, not by preference:
|
|
1975
|
+
|
|
1976
|
+
| Form | For a type that |
|
|
1977
|
+
| ---- | --------------- |
|
|
1978
|
+
| `from_file` | opens the media itself, bringing its frame or sample rate along (MiniMax-H3's references) |
|
|
1979
|
+
| `from_previous_result` | declares a media `kind`, so a step's output lands in the right field on its own |
|
|
1980
|
+
| `from_arguments` | is a plain record of fields - no `from_file()`, no `kind` (LTX-2's conditions and references) |
|
|
1981
|
+
|
|
1982
|
+
See [workflows/templates/ltx2/keyframes.json](../workflows/templates/ltx2/keyframes.json) for the file form and
|
|
1983
|
+
[workflows/templates/ltx2/extend-clip.json](../workflows/templates/ltx2/extend-clip.json) for the one built from an
|
|
1984
|
+
earlier step.
|
|
1985
|
+
|
|
1986
|
+
### Frames Across a Step Boundary
|
|
1987
|
+
|
|
1988
|
+
A pipeline that generates video with a soundtrack returns the two paired, and the result
|
|
1989
|
+
muxes them into one file. A step that works on the frames alone - a latent upsampler, an
|
|
1990
|
+
interpolator - returns frames without it. Two tasks carry the pieces across:
|
|
1991
|
+
|
|
1992
|
+
- **`video_frames`** takes a generated video and returns its frames as one
|
|
1993
|
+
`(frames, height, width, channels)` uint8 array - the 0-255 shape LTX-2's conditions
|
|
1994
|
+
want, and one artifact rather than one per frame.
|
|
1995
|
+
- **`pair_audio`** puts a soundtrack back beside frames that lost it, so the step that
|
|
1996
|
+
saves them writes a single muxed mp4.
|
|
1997
|
+
|
|
1998
|
+
```json
|
|
1999
|
+
{
|
|
2000
|
+
"name": "film",
|
|
2001
|
+
"task": {
|
|
2002
|
+
"command": "pair_audio",
|
|
2003
|
+
"arguments": {
|
|
2004
|
+
"video": "previous_result:edit",
|
|
2005
|
+
"audio": "previous_result:balanced",
|
|
2006
|
+
"sample_rate": "variable:sample_rate"
|
|
2007
|
+
}
|
|
2008
|
+
},
|
|
2009
|
+
"result": { "content_type": "video/mp4", "fps": 24 }
|
|
2010
|
+
}
|
|
2011
|
+
```
|
|
2012
|
+
|
|
2013
|
+
`audio` takes either a waveform or the earlier step whose video carried the soundtrack,
|
|
2014
|
+
which brings its sample rate along; here it is an earlier step's waveform, so
|
|
2015
|
+
`sample_rate` is given explicitly. The frames keep the rate they arrived
|
|
2016
|
+
with - `video` given a file or an `asset:` carries that file's fps through
|
|
2017
|
+
to the saved mp4 - so `result.fps` is only needed for frames that bring no
|
|
2018
|
+
rate of their own. A mono track needs no preparation: an mp4 audio stream
|
|
2019
|
+
takes stereo and nothing else, so saving duplicates the single channel into
|
|
2020
|
+
two and emits a warning saying it did.
|
|
2021
|
+
|
|
2022
|
+
The track and the frames are two lengths a workflow used to have to keep equal by
|
|
2023
|
+
hand. `"fit": "video"` derives one from the other instead: the track is cut to
|
|
2024
|
+
exactly the frames it is laid over, or padded with silence and warned about when it
|
|
2025
|
+
is shorter than they are. That is what a soundtrack over a cut whose length is an
|
|
2026
|
+
argument needs - nothing in a workflow can multiply a list's length by a frame
|
|
2027
|
+
count, so `music-video.json` sliced a fixed 496 frames of song while its cut
|
|
2028
|
+
followed a `shots` list, and a two-shot run wrote 10.3 s of picture into a 20.7 s
|
|
2029
|
+
container and reported `succeeded` with no warnings (#142). Left unset the track is
|
|
2030
|
+
used as it is and a disagreement is warned about rather than passing in silence.
|
|
2031
|
+
|
|
2032
|
+
Which shape a pipeline argument wants is the pipeline's business, and the two LTX-2
|
|
2033
|
+
paths differ: a keyframe condition is mapped from 0-255, so it takes the `video_frames`
|
|
2034
|
+
array, while an IC-LoRA reference goes through the video processor, which expects the
|
|
2035
|
+
`[0, 1]` frames the pipeline returned - `previous_result:step.frames` hands those over
|
|
2036
|
+
untouched.
|
|
2037
|
+
|
|
2038
|
+
**Example:** [workflows/templates/assemble-and-score.json](../workflows/templates/assemble-and-score.json)
|