diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/docs/TASKS.md
ADDED
|
@@ -0,0 +1,1741 @@
|
|
|
1
|
+
# Task Commands
|
|
2
|
+
|
|
3
|
+
Tasks are utility operations that run outside of pipeline inference. Use them for image preprocessing, data gathering, and other non-model operations.
|
|
4
|
+
|
|
5
|
+
```json
|
|
6
|
+
{
|
|
7
|
+
"name": "step_name",
|
|
8
|
+
"task": {
|
|
9
|
+
"command": "command_name",
|
|
10
|
+
"arguments": { ... }
|
|
11
|
+
},
|
|
12
|
+
"result": { "content_type": "image/jpeg" }
|
|
13
|
+
}
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Any task that runs a model accepts a `"device"` argument to pin where it runs -
|
|
17
|
+
useful for keeping a helper model (a captioner, an upscaler) off the accelerator a
|
|
18
|
+
loaded pipeline is using, or on a second one. `"device"` is listed on every task's
|
|
19
|
+
schema, since it is always safe to pass: a task that runs no model - `slice_audio`,
|
|
20
|
+
`compose_text`, and the like - just ignores it.
|
|
21
|
+
|
|
22
|
+
Task argument schemas are discoverable: `GET /api/tasks/{command}` on the
|
|
23
|
+
[server](SERVER.md) returns each command's arguments read from its registered
|
|
24
|
+
implementation's real signature, the web editor builds task forms from them,
|
|
25
|
+
and workflow validation flags task-argument typos the same way it flags
|
|
26
|
+
pipeline ones.
|
|
27
|
+
|
|
28
|
+
A signature carries no domain, though, so the numbers whose domain is not a
|
|
29
|
+
judgement call are declared separately (`dw/task_domains.py`) and validation
|
|
30
|
+
reports one outside it as an error at its JSON path: a count of frames or
|
|
31
|
+
seconds to cut, and a sample rate or frame rate, have to be above zero, and an
|
|
32
|
+
offset to start at zero or above. Those are refused rather than interpreted -
|
|
33
|
+
`num_frames: -10` used to answer with the track minus its last ten frames and
|
|
34
|
+
`target_sample_rate: 0` with the original samples under a 44100 Hz header, both
|
|
35
|
+
reported as clean successes. The commands refuse the same values at run time,
|
|
36
|
+
which is what catches one that arrived from a `variable:` or an earlier step
|
|
37
|
+
rather than being written in the file.
|
|
38
|
+
|
|
39
|
+
## Image Processing
|
|
40
|
+
|
|
41
|
+
### ControlNet Preprocessors
|
|
42
|
+
|
|
43
|
+
Generate control images for ControlNet pipelines:
|
|
44
|
+
|
|
45
|
+
| Command | Description |
|
|
46
|
+
| ------- | ----------- |
|
|
47
|
+
| `canny` | Canny edge detection |
|
|
48
|
+
| `canny_cv` | OpenCV Canny (alternative) |
|
|
49
|
+
| `depth` | Depth estimation (DPT) |
|
|
50
|
+
| `midas` | Monocular depth (MiDaS) |
|
|
51
|
+
| `zoe` | Zoe depth estimation |
|
|
52
|
+
| `zoe_depth` | Zoe depth with colorization |
|
|
53
|
+
| `leres` | Relative depth (LeReS) |
|
|
54
|
+
| `normal_bae` | Surface normal estimation |
|
|
55
|
+
| `openpose` | Pose estimation |
|
|
56
|
+
| `dw_pose` | DW pose estimation |
|
|
57
|
+
| `mlsd` | Line segment detection |
|
|
58
|
+
| `lineart` | Line art extraction |
|
|
59
|
+
| `lineart_standard` | Standard line art |
|
|
60
|
+
| `hed` | HED edge detection |
|
|
61
|
+
| `scribble` | Scribble-style edges |
|
|
62
|
+
| `pidi` | Boundary detection |
|
|
63
|
+
| `shuffle` | Content-preserving shuffle |
|
|
64
|
+
| `teed` | TEED edge detection |
|
|
65
|
+
| `anyline` | Anyline edge detection |
|
|
66
|
+
| `sam` | Segment Anything |
|
|
67
|
+
| `segmentation` | Semantic segmentation |
|
|
68
|
+
| `depth_estimator` | Depth hint generation |
|
|
69
|
+
| `depth_estimator_tensor` | Depth hint as tensor |
|
|
70
|
+
|
|
71
|
+
All accept an `image` argument with processing parameters:
|
|
72
|
+
|
|
73
|
+
```json
|
|
74
|
+
{
|
|
75
|
+
"task": {
|
|
76
|
+
"command": "canny",
|
|
77
|
+
"arguments": {
|
|
78
|
+
"image": {
|
|
79
|
+
"location": "https://example.com/photo.jpg",
|
|
80
|
+
"low_threshold": 50,
|
|
81
|
+
"high_threshold": 200,
|
|
82
|
+
"detect_resolution": 1024,
|
|
83
|
+
"image_resolution": 1024
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### Image Manipulation
|
|
91
|
+
|
|
92
|
+
| Command | Description | Extra Arguments |
|
|
93
|
+
| ------- | ----------- | --------------- |
|
|
94
|
+
| `remove_background` | Remove image background | |
|
|
95
|
+
| `resize_center_crop` | Resize with center crop | `width`, `height` |
|
|
96
|
+
| `resize_resample` | Resample to nearest 64px multiple | |
|
|
97
|
+
| `resize_rescale` | Resize to exact dimensions | `width`, `height` |
|
|
98
|
+
| `resize_bucket` | Snap to closest model-native aspect ratio | `resolution`, `ratios`, `alignment` |
|
|
99
|
+
| `crop_square` | Center crop to square | |
|
|
100
|
+
| `recenter_crop` | Re-frame around a chosen point at a chosen scale, so a series of images registers on one feature; the window may run off the source | `center_x`, `center_y`, `crop`, `width`, `height`, `fill` |
|
|
101
|
+
| `add_border_and_mask` | Add border with alpha mask | |
|
|
102
|
+
| `add_border_and_mask_with_size` | Border with specific dimensions | `width`, `height` |
|
|
103
|
+
| `strip_exif` | Remove all EXIF/metadata from image | |
|
|
104
|
+
| `add_watermark` | Add visible text watermark | `text`, `position`, `opacity`, `font_size`, `color`, `margin` |
|
|
105
|
+
| `get_image_size` | Return `{width, height}` dict | |
|
|
106
|
+
|
|
107
|
+
### EXIF Stripping
|
|
108
|
+
|
|
109
|
+
Remove all EXIF metadata, GPS coordinates, camera info, and timestamps from images for privacy-safe preprocessing:
|
|
110
|
+
|
|
111
|
+
```json
|
|
112
|
+
{
|
|
113
|
+
"task": {
|
|
114
|
+
"command": "strip_exif",
|
|
115
|
+
"arguments": {
|
|
116
|
+
"image": "previous_result:input_image"
|
|
117
|
+
}
|
|
118
|
+
},
|
|
119
|
+
"result": { "content_type": "image/png" }
|
|
120
|
+
}
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
Returns a clean copy with pixel data only — no embedded metadata. Useful as a first step when processing user-uploaded images.
|
|
124
|
+
|
|
125
|
+
### Watermark Embedding
|
|
126
|
+
|
|
127
|
+
Add a visible text watermark to images for responsible AI compliance:
|
|
128
|
+
|
|
129
|
+
```json
|
|
130
|
+
{
|
|
131
|
+
"task": {
|
|
132
|
+
"command": "add_watermark",
|
|
133
|
+
"arguments": {
|
|
134
|
+
"image": "previous_result:generate",
|
|
135
|
+
"text": "AI Generated",
|
|
136
|
+
"position": "bottom-right",
|
|
137
|
+
"opacity": 128
|
|
138
|
+
}
|
|
139
|
+
},
|
|
140
|
+
"result": { "content_type": "image/png" }
|
|
141
|
+
}
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
| Argument | Required | Description |
|
|
145
|
+
| -------- | -------- | ----------- |
|
|
146
|
+
| `text` | No | Watermark text (default: "AI Generated") |
|
|
147
|
+
| `position` | No | "bottom-right", "bottom-left", "top-right", "top-left", or "center" (default: "bottom-right") |
|
|
148
|
+
| `opacity` | No | Text opacity 0-255 (default: 128) |
|
|
149
|
+
| `font_size` | No | Font size in pixels, 0 = auto-scale ~3% of image height (default: 0) |
|
|
150
|
+
| `color` | No | RGB array for text color (default: white) |
|
|
151
|
+
| `margin` | No | Pixel margin from edges (default: 10) |
|
|
152
|
+
|
|
153
|
+
### Aspect Ratio Bucketing
|
|
154
|
+
|
|
155
|
+
The `resize_bucket` command snaps an image to the closest model-native aspect ratio, then resizes with 64-pixel alignment. This avoids distortion and ensures the model generates at a resolution it was trained on.
|
|
156
|
+
|
|
157
|
+
```json
|
|
158
|
+
{
|
|
159
|
+
"task": {
|
|
160
|
+
"command": "resize_bucket",
|
|
161
|
+
"arguments": {
|
|
162
|
+
"image": "previous_result:input_image",
|
|
163
|
+
"resolution": 1024
|
|
164
|
+
}
|
|
165
|
+
},
|
|
166
|
+
"result": { "content_type": "image/png" }
|
|
167
|
+
}
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
| Argument | Required | Description |
|
|
171
|
+
| -------- | -------- | ----------- |
|
|
172
|
+
| `resolution` | No | Target short-side size in pixels (default: 1024) |
|
|
173
|
+
| `ratios` | No | Custom list of `[w, h]` ratio pairs (default: standard SDXL/Flux ratios) |
|
|
174
|
+
| `alignment` | No | Round dimensions to this multiple (default: 64) |
|
|
175
|
+
|
|
176
|
+
**Default ratios:** 1:1, 4:3, 3:4, 3:2, 2:3, 16:9, 9:16, 21:9, 9:21
|
|
177
|
+
|
|
178
|
+
For example, a 1600x900 photo (16:9) at resolution 1024 becomes 1792x1024. A 800x600 photo (4:3) becomes 1344x1024.
|
|
179
|
+
|
|
180
|
+
## Video Processing
|
|
181
|
+
|
|
182
|
+
| Command | Description | Extra Arguments |
|
|
183
|
+
| ------- | ----------- | --------------- |
|
|
184
|
+
| `get_first_frame` | Extract first video frame | |
|
|
185
|
+
| `get_last_frame` | Extract last video frame | |
|
|
186
|
+
| `get_frame` | Extract frame at index | `frame_index` |
|
|
187
|
+
|
|
188
|
+
The frame commands accept videos in any shape a result carries them: PIL frame
|
|
189
|
+
lists, numpy or torch frame arrays, and audio+video pairs (LTX-2, MiniMax H3).
|
|
190
|
+
The extracted frame is always a PIL image.
|
|
191
|
+
|
|
192
|
+
### concat_videos
|
|
193
|
+
|
|
194
|
+
Concatenate videos - and the audio generated with them - into one video. The
|
|
195
|
+
standalone counterpart of a chained pipeline step's stitching (see "Chained
|
|
196
|
+
video generation" in the workflow guide):
|
|
197
|
+
|
|
198
|
+
```json
|
|
199
|
+
{
|
|
200
|
+
"task": {
|
|
201
|
+
"command": "concat_videos",
|
|
202
|
+
"arguments": {
|
|
203
|
+
"videos": ["previous_result:shot_1", "previous_result:shot_2"],
|
|
204
|
+
"trim_frames": 1,
|
|
205
|
+
"crossfade_ms": 75,
|
|
206
|
+
"fps": 24
|
|
207
|
+
}
|
|
208
|
+
},
|
|
209
|
+
"result": { "content_type": "video/mp4", "fps": 24 }
|
|
210
|
+
}
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
| Argument | Required | Description |
|
|
214
|
+
| -------- | -------- | ----------- |
|
|
215
|
+
| `videos` | Yes | The videos to join, in order - `previous_result` references, or the path or URL of a video file an earlier run wrote, which is read with the audio muxed into it |
|
|
216
|
+
| `trim_frames` | No | Frames dropped from the head of every video after the first (default: 0) |
|
|
217
|
+
| `crossfade_ms` | No | Equal-power crossfade at each audio seam, drawn from the trimmed material - no effect when `trim_frames` is 0, and validation warns when one is written there (default: 75) |
|
|
218
|
+
| `audio_bleed_ms` | No | How long the outgoing video's tail rings on over the head of the next one, at seams with nothing trimmed to crossfade (default: 0, off) |
|
|
219
|
+
| `audio_bleed_gain_db` | No | Gain applied to the bled tail before it is added, in dB - negative ducks a tail that would otherwise push the seam over 0 dBFS (default: 0, unchanged) |
|
|
220
|
+
| `seam_fade_ms` | No | Fade on each side of a seam that gets neither a crossfade nor a bleed - for tonal material, not for a continuous bed (default: 3, just enough not to click) |
|
|
221
|
+
| `fps` | No | Frame rate of the videos - required to join audio when trimming, and the rate the joined file is written at unless `result.fps` overrides it |
|
|
222
|
+
| `match_levels` | No | Even the shots' loudness out before joining - `"rms"` for perceived level (the measurement `get_gallery_metadata` reports as `mean_dbfs`), `"peak"` for the loudest sample. Off by default |
|
|
223
|
+
| `match_levels_dbfs` | No | The level `match_levels` moves every shot to (default: -1 dBFS for `peak`, -20 dBFS for `rms`). A shot that would clip at the target is held at -0.5 dBFS peak instead, reported as a `match_levels_held` warning with a per-shot log event |
|
|
224
|
+
|
|
225
|
+
A video may also be named by path or URL, which is how shots an earlier run
|
|
226
|
+
already wrote are joined without regenerating them - the file is read with the
|
|
227
|
+
audio muxed into it, and its track is fitted to the frames' own duration so the
|
|
228
|
+
codec's block padding does not walk the sound off the picture over a dozen
|
|
229
|
+
seams. A shot generated in memory through a `previous_result:` chain gets the
|
|
230
|
+
same fit, applied where the file is written rather than where it is decoded,
|
|
231
|
+
so per-shot drift does not accumulate across a cut the way it once did:
|
|
232
|
+
|
|
233
|
+
```json
|
|
234
|
+
{
|
|
235
|
+
"task": {
|
|
236
|
+
"command": "concat_videos",
|
|
237
|
+
"arguments": {
|
|
238
|
+
"videos": [
|
|
239
|
+
"/path/to/outputs/shot_01.mp4",
|
|
240
|
+
"/path/to/outputs/shot_02.mp4",
|
|
241
|
+
"previous_result:shot_03_rerendered"
|
|
242
|
+
],
|
|
243
|
+
"trim_frames": 0,
|
|
244
|
+
"fps": 24
|
|
245
|
+
}
|
|
246
|
+
},
|
|
247
|
+
"result": { "content_type": "video/mp4", "fps": 24 }
|
|
248
|
+
}
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
Give each video its own entry. One `previous_result` reference naming a step
|
|
252
|
+
that produced several videos does not hand them all over at once - it fans the
|
|
253
|
+
step out over them, one concatenation per video, which is what makes the list
|
|
254
|
+
form above the way to join a run's shots.
|
|
255
|
+
|
|
256
|
+
`trim_frames` and `audio_bleed_ms` address opposite situations. A *chain* carries
|
|
257
|
+
its keyframe forward, so the trimmed head is material that covers the same stretch
|
|
258
|
+
of time as the outgoing tail and the two can be crossfaded. A *cut* generates each
|
|
259
|
+
shot independently, so there is nothing to fade with - and generated shots tend to
|
|
260
|
+
open on near-silence and end mid-sound, leaving a butt-join that drops a running
|
|
261
|
+
laugh track or a ringing room into a hole. `audio_bleed_ms` fills it the way an
|
|
262
|
+
audience carries across a picture cut: a decaying copy of the outgoing tail is laid
|
|
263
|
+
over the incoming head, added to whatever is already there, shortening neither side.
|
|
264
|
+
Reach for more than the gap looks like it needs: a shot's head is silent for
|
|
265
|
+
longer than the picture suggests, and the bleed has to outlast it. Measured on
|
|
266
|
+
a five-shot H3 sitcom cut, 700 ms still left a 44 dB hole at the worst seam;
|
|
267
|
+
1800 ms brought it to 32 dB and 2500 ms gained almost nothing more, so the
|
|
268
|
+
`dialogue-short` template defaults to 1800 and exposes it as `audio_bleed_ms`:
|
|
269
|
+
|
|
270
|
+
```json
|
|
271
|
+
{
|
|
272
|
+
"task": {
|
|
273
|
+
"command": "concat_videos",
|
|
274
|
+
"arguments": {
|
|
275
|
+
"videos": ["previous_result:shot_1", "previous_result:shot_2"],
|
|
276
|
+
"trim_frames": 0,
|
|
277
|
+
"audio_bleed_ms": 1800,
|
|
278
|
+
"fps": 24
|
|
279
|
+
}
|
|
280
|
+
},
|
|
281
|
+
"result": { "content_type": "video/mp4", "fps": 24 }
|
|
282
|
+
}
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
A bleed works because it copies ambience, which has no pitch and no attacks to
|
|
286
|
+
give the copy away. It is the wrong tool for anything tonal - a copied musical
|
|
287
|
+
phrase or half-spoken word reads as a stutter whichever direction it runs.
|
|
288
|
+
`bleed_join` checks the outgoing tail's spectral flatness and warns when it
|
|
289
|
+
looks tonal or speech-like rather than noise-like, so this failure mode
|
|
290
|
+
surfaces in the job's `warnings` list instead of only in the mix. When a
|
|
291
|
+
shot ends on something tonal, either give the cut a continuous bed with
|
|
292
|
+
`slice_audio` + `loop_audio` + `mix_audio` + `pair_audio`, which leaves no seam
|
|
293
|
+
to treat at all, or fade the
|
|
294
|
+
seam gracefully with `seam_fade_ms` (a hundred or so milliseconds) and accept the
|
|
295
|
+
cut. That advice inverts on a continuous bed - a laugh track, room tone - where a
|
|
296
|
+
longer fade only digs the hole deeper (the same sitcom cut measured 54-59 dB
|
|
297
|
+
holes with a 250-500 ms fade and no bleed). `audio_bleed_ms` wins where both are
|
|
298
|
+
set and there is material to bleed. A bleed covers the gap but cannot fill it:
|
|
299
|
+
the silence is inside the incoming shot's own head, and the only complete fix is
|
|
300
|
+
a continuous bed under the whole cut: `slice_audio` a few seconds of tone out of
|
|
301
|
+
a shot, [`loop_audio`](#loop_audio) it to the length of the cut, `mix_audio` it
|
|
302
|
+
under the episode and `pair_audio` it back onto the picture.
|
|
303
|
+
|
|
304
|
+
Levels are the other thing a cut has to reconcile, and no fade control can
|
|
305
|
+
touch it. Shots generated independently land wherever the model put them - two
|
|
306
|
+
shots of one scene, same template, same cast, measured `peak_dbfs` -2.65 and
|
|
307
|
+
-12.56 - and each reads as fine on its own, because a shot is only wrong
|
|
308
|
+
*relative to what it is cut against*. Butt-joined, that is a 10 dB drop at the
|
|
309
|
+
cut, and it is not an artifact *at* the seam that a fade could smooth: it is
|
|
310
|
+
either side of it. `match_levels` scales each track before the join -
|
|
311
|
+
`"rms"` matches perceived level, which is usually what "make these sound the
|
|
312
|
+
same" means, and `"peak"` matches the loudest sample, which is the safer
|
|
313
|
+
choice on material with big transients. A shot whose gain would clip at the
|
|
314
|
+
target is held just below full scale and the log says so. Left off - the
|
|
315
|
+
default, so nothing existing changes - a spread of 6 dB or more across the
|
|
316
|
+
tracks being joined is reported as a warning rather than passing in silence:
|
|
317
|
+
on the job's `warnings` and as a `warning` event in its stream, not only in
|
|
318
|
+
the server's log, since the caller who can act on it is the one who asked for
|
|
319
|
+
the run. The other end of the range warns too: a shot at or below -40 dBFS,
|
|
320
|
+
or one that needs 20 dB or more of gain to reach the target, is noise floor
|
|
321
|
+
rather than a quieter performance, and matching it up is reported as
|
|
322
|
+
`match_levels_near_silent`. `dissolve_videos` takes the same pair.
|
|
323
|
+
|
|
324
|
+
The joined soundtrack is fitted to the joined frames. A track that comes out
|
|
325
|
+
short of the frame grid - rounding in an input's own track, which otherwise
|
|
326
|
+
compounds join after join - is padded with silence to it. A pad of a frame
|
|
327
|
+
or more is a warning (`joined_audio_padded_to_frames`); less than a frame is
|
|
328
|
+
rounding, and only logged. The file's AAC encode can then trim the track by a
|
|
329
|
+
further handful of samples (typically 16-32, under a millisecond), which is
|
|
330
|
+
logged the same way, or warned as `joined_audio_short_after_mux` if it
|
|
331
|
+
reaches a frame. Either way the recorded shots are re-measured against the
|
|
332
|
+
file as written, so `media.shots` stays accurate. Both apply to
|
|
333
|
+
`dissolve_videos` the same way, and neither task warns about resampling
|
|
334
|
+
inputs that already agree to a `sample_rate` the caller pinned.
|
|
335
|
+
|
|
336
|
+
### dissolve_videos
|
|
337
|
+
|
|
338
|
+
Join videos with a cross-dissolve at every seam, and fade the whole piece in
|
|
339
|
+
from and out to a colour. Where `concat_videos` cuts - right for shots that
|
|
340
|
+
each carry their own sound - this melts one shot into the next, which is what a
|
|
341
|
+
montage cut to a score wants:
|
|
342
|
+
|
|
343
|
+
```json
|
|
344
|
+
{
|
|
345
|
+
"task": {
|
|
346
|
+
"command": "dissolve_videos",
|
|
347
|
+
"arguments": {
|
|
348
|
+
"videos": ["previous_result:shot_1", "previous_result:shot_2"],
|
|
349
|
+
"dissolve_frames": 12,
|
|
350
|
+
"fade_in_frames": 12,
|
|
351
|
+
"fade_out_frames": 24,
|
|
352
|
+
"fps": 24
|
|
353
|
+
}
|
|
354
|
+
},
|
|
355
|
+
"result": { "content_type": "video/mp4", "fps": 24 }
|
|
356
|
+
}
|
|
357
|
+
```
|
|
358
|
+
|
|
359
|
+
| Argument | Required | Description |
|
|
360
|
+
| -------- | -------- | ----------- |
|
|
361
|
+
| `videos` | Yes | The videos to join, in order - `previous_result` references, or the path or URL of a video file an earlier run wrote, one entry per video as with `concat_videos` |
|
|
362
|
+
| `dissolve_frames` | No | Frames of overlap at each seam, blended linearly (default: 12). 0 is a hard cut |
|
|
363
|
+
| `fade_in_frames` | No | Frames over which the first video rises out of `fade_color` (default: 0) |
|
|
364
|
+
| `fade_out_frames` | No | Frames over which the last video sinks into it (default: 0) |
|
|
365
|
+
| `fade_color` | No | The RGB colour the fades come from and go to (default: black) |
|
|
366
|
+
| `fps` | No | Frame rate of the videos - required to crossfade audio at a dissolve, and the rate the dissolved file is written at unless `result.fps` overrides it |
|
|
367
|
+
| `match_levels` | No | Even the shots' loudness out before joining - `"rms"` or `"peak"`, as with [`concat_videos`](#concat_videos). Off by default |
|
|
368
|
+
| `match_levels_dbfs` | No | The level `match_levels` moves every shot to (default: -1 dBFS for `peak`, -20 dBFS for `rms`). A shot that would clip at the target is held at -0.5 dBFS peak instead, reported as a `match_levels_held` warning with a per-shot log event |
|
|
369
|
+
|
|
370
|
+
Every seam shortens the result by one overlap, so eight 124-frame shots joined
|
|
371
|
+
with 12-frame dissolves run 908 frames, not 992 - size a soundtrack slice to
|
|
372
|
+
the joined length, not the sum. When every input carries audio, the tracks are
|
|
373
|
+
crossfaded over exactly the seam's span so they stay in step with the picture;
|
|
374
|
+
when any input is silent the result is, and `pair_audio` puts a score under it.
|
|
375
|
+
|
|
376
|
+
**Example:** [dissolve-between-shots.json](../workflows/templates/dissolve-between-shots.json)
|
|
377
|
+
|
|
378
|
+
### stabilize_video
|
|
379
|
+
|
|
380
|
+
Remove a generated clip's accumulated framing drift - the slow wander a video
|
|
381
|
+
model adds over a shot that was meant to hold still:
|
|
382
|
+
|
|
383
|
+
```json
|
|
384
|
+
{
|
|
385
|
+
"task": {
|
|
386
|
+
"command": "stabilize_video",
|
|
387
|
+
"arguments": {
|
|
388
|
+
"clip": "variable:shot_1",
|
|
389
|
+
"smooth": 0
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
```
|
|
394
|
+
|
|
395
|
+
| Argument | Required | Description |
|
|
396
|
+
| -------- | -------- | ----------- |
|
|
397
|
+
| `clip` | Yes | The video - a frame list, a frame array or tensor, an audio+video pair, or the path or URL of a video file, read with its audio, so a shot an earlier run wrote can be steadied without regenerating it |
|
|
398
|
+
| `smooth` | No | `0` (the default) locks the framing to the first frame, which is what a shot generated from a pinned keyframe wants. A window in frames instead removes only the wander faster than that window, so a slow deliberate camera move survives and the drift around it does not |
|
|
399
|
+
|
|
400
|
+
The argument is `clip`, not `video`, on purpose: the engine loads an argument
|
|
401
|
+
named `video` itself, as bare frames, which would strip the soundtrack off
|
|
402
|
+
before the task ever saw it. Frames are shifted back and the result is cropped
|
|
403
|
+
to the region every frame covers, then resized to the original size; a
|
|
404
|
+
soundtrack passes through untouched.
|
|
405
|
+
|
|
406
|
+
It is a stabilization pass, not a format pass. `smooth: 0` on a shot with a
|
|
407
|
+
deliberate camera move fights the move - every frame is shifted back toward
|
|
408
|
+
the first, cropped and rescaled - and nothing downstream will notice, since
|
|
409
|
+
the duration, size and sample rate all survive. Run it on a shot that drifts,
|
|
410
|
+
as its own step; do not run it on every shot before a cut, which is what the
|
|
411
|
+
assembly templates once did and what made their output visibly wider than
|
|
412
|
+
the source. The join tasks refuse shots of different sizes, so no
|
|
413
|
+
normalization step is needed before them.
|
|
414
|
+
|
|
415
|
+
### video_frames
|
|
416
|
+
|
|
417
|
+
The frames of a generated video, as one `(frames, height, width, channels)`
|
|
418
|
+
uint8 array. That is the shape an argument taking frames rather than a video
|
|
419
|
+
wants - LTX-2's keyframe conditions, which are mapped from 0-255 - and it is one
|
|
420
|
+
artifact where a list of frames would become one artifact per frame and multiply
|
|
421
|
+
the step that consumed it:
|
|
422
|
+
|
|
423
|
+
```json
|
|
424
|
+
{
|
|
425
|
+
"name": "opening_frames",
|
|
426
|
+
"task": {
|
|
427
|
+
"command": "video_frames",
|
|
428
|
+
"arguments": { "video": "previous_result:opening" }
|
|
429
|
+
},
|
|
430
|
+
"result": { "content_type": "video/mp4", "save": false, "fps": 24 }
|
|
431
|
+
}
|
|
432
|
+
```
|
|
433
|
+
|
|
434
|
+
| Argument | Required | Description |
|
|
435
|
+
| -------- | -------- | ----------- |
|
|
436
|
+
| `video` | Yes | The video - a frame list, a frame array or tensor, or an audio+video pair |
|
|
437
|
+
|
|
438
|
+
An argument that goes through diffusers' video processor instead - LTX-2's
|
|
439
|
+
IC-LoRA references - wants the `[0, 1]` frames the pipeline returned rather than
|
|
440
|
+
this array; hand those over with `previous_result:step.frames`.
|
|
441
|
+
|
|
442
|
+
**Example:** [extend-clip.json](../workflows/templates/ltx2/extend-clip.json)
|
|
443
|
+
|
|
444
|
+
### pair_audio
|
|
445
|
+
|
|
446
|
+
Pair a video with an audio track, so the two are saved as one muxed file. A
|
|
447
|
+
pipeline that generates its own soundtrack returns the pair together; anything
|
|
448
|
+
working on the frames alone - a latent upsampler, an interpolator, an upscaler -
|
|
449
|
+
returns frames without it, and this puts it back:
|
|
450
|
+
|
|
451
|
+
```json
|
|
452
|
+
{
|
|
453
|
+
"task": {
|
|
454
|
+
"command": "pair_audio",
|
|
455
|
+
"arguments": {
|
|
456
|
+
"video": "previous_result:upscale",
|
|
457
|
+
"audio": "previous_result:base"
|
|
458
|
+
}
|
|
459
|
+
},
|
|
460
|
+
"result": { "content_type": "video/mp4", "fps": 24 }
|
|
461
|
+
}
|
|
462
|
+
```
|
|
463
|
+
|
|
464
|
+
| Argument | Required | Description |
|
|
465
|
+
| -------- | -------- | ----------- |
|
|
466
|
+
| `video` | Yes | The frames - a frame list, a frame array or tensor, or an audio+video pair whose own soundtrack is replaced; their own rate is carried through to the output, so `result.fps` is only needed to override it (frames that carry none are written at 8 fps) |
|
|
467
|
+
| `audio` | Yes | The soundtrack - a waveform, the earlier step whose video carried one, or the path or URL of an audio or video file; the last two bring their sample rate along. A mono track is fine: an mp4 audio stream takes stereo and nothing else, so saving duplicates the one channel into two and warns that it did |
|
|
468
|
+
| `sample_rate` | No | Sample rate of the waveform. Required unless `audio` carries one; given here it wins |
|
|
469
|
+
| `fps` | No | The rate the frames play at, only needed when they carry none of their own. Used solely to work out how long the video is - what `fit` and the length-mismatch check measure the track against - and is never written to the file; that's `result.fps`, which sets the rate the output plays at and defaults to 8 fps when the frames carry none |
|
|
470
|
+
| `fit` | No | `"video"` cuts or pads the track with silence to the length of the frames, warning either way (`audio_padded_to_video` / `audio_trimmed_to_video`). Left unset (the default) the track is used as it is, and a length that disagrees with the frames' is warned about rather than corrected (`audio_video_length_mismatch`). Any other value is refused at run time, not by `validate_workflow` |
|
|
471
|
+
|
|
472
|
+
`fit`'s guarantee is exact for the waveform handed to the encoder, not for
|
|
473
|
+
the file the encoder writes: muxing is a lossy AAC encode, and it can still
|
|
474
|
+
trim or pad the written track by a further handful of samples (#428
|
|
475
|
+
measured up to ~30, under a millisecond). That residual is logged, not
|
|
476
|
+
warned; on a video with recorded `shots` it becomes a
|
|
477
|
+
`joined_audio_short_after_mux` warning only if it reaches a frame. `get_gallery_metadata`'s
|
|
478
|
+
`media.shots` and `assess_output`'s `sync_length` are measured against the
|
|
479
|
+
written file, not the pre-encode prediction, so they are the number to
|
|
480
|
+
trust for the track's actual length.
|
|
481
|
+
|
|
482
|
+
When the video carries recorded `shots` (from an earlier `concat_videos`,
|
|
483
|
+
`dissolve_videos` or chain step), `pair_audio` remeasures each one's sample
|
|
484
|
+
fields against the track it was handed. Every shot but the last is
|
|
485
|
+
`round(start_frame / fps * sample_rate)`; the last one runs to the track's
|
|
486
|
+
actual end, and once the file is written it is measured again against what
|
|
487
|
+
the file decodes to. So its `num_samples` can sit a few dozen samples off
|
|
488
|
+
`round(num_frames * sample_rate / fps)`: the encoder's trim, which the job's
|
|
489
|
+
event log records. A real mismatch between the
|
|
490
|
+
track and the video's length is a separate, thresholded warning
|
|
491
|
+
(`audio_video_length_mismatch`, or `audio_padded_to_video` /
|
|
492
|
+
`audio_trimmed_to_video` when `fit: "video"` corrected it), so a last shot
|
|
493
|
+
short by less than a millisecond is expected, not a bug. `get_gallery_metadata`'s `media.shots`
|
|
494
|
+
reports the remeasured fields.
|
|
495
|
+
|
|
496
|
+
**Example:** [assemble-and-score.json](../workflows/templates/assemble-and-score.json)
|
|
497
|
+
|
|
498
|
+
### slice_audio
|
|
499
|
+
|
|
500
|
+
Cut a slice out of an audio track, addressed in seconds or in video frames.
|
|
501
|
+
Slices reaching past the end of the track are zero-padded — asking for more
|
|
502
|
+
than the source holds returns a track of the length you asked for whose tail is
|
|
503
|
+
digital silence, not a shorter track and not an error. Anything past a few
|
|
504
|
+
milliseconds of that padding is reported as a `slice_past_end` warning on the
|
|
505
|
+
job, because a score laid under a longer cut goes silent for the rest of the
|
|
506
|
+
film without anything else saying so; to fill a cut longer than the recording,
|
|
507
|
+
build a bed with [`loop_audio`](#loop_audio) first and slice that. Either half
|
|
508
|
+
of a pair may be left out - an omitted start begins at the head of the track, an omitted
|
|
509
|
+
duration runs to the end of it - so a workflow that trims only when it is given
|
|
510
|
+
a length still passes the whole track along:
|
|
511
|
+
|
|
512
|
+
```json
|
|
513
|
+
{
|
|
514
|
+
"task": {
|
|
515
|
+
"command": "slice_audio",
|
|
516
|
+
"arguments": {
|
|
517
|
+
"audio": "./soundtrack.wav",
|
|
518
|
+
"start_frame": 124,
|
|
519
|
+
"num_frames": 124,
|
|
520
|
+
"fps": 24
|
|
521
|
+
}
|
|
522
|
+
},
|
|
523
|
+
"result": { "content_type": "audio/wav", "sample_rate": 44100 }
|
|
524
|
+
}
|
|
525
|
+
```
|
|
526
|
+
|
|
527
|
+
| Argument | Required | Description |
|
|
528
|
+
| -------- | -------- | ----------- |
|
|
529
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a waveform from a previous step, or an earlier step's video generated with a soundtrack (which brings its sample rate along) |
|
|
530
|
+
| `start_seconds` / `duration_seconds` | One pair | The slice in seconds; either may be omitted |
|
|
531
|
+
| `start_frame` / `num_frames` / `fps` | One pair | The slice in video frames; `fps` is required, start and count may be omitted |
|
|
532
|
+
| `sample_rate` | With a waveform | Sample rate of a directly passed waveform (files carry their own) |
|
|
533
|
+
|
|
534
|
+
### gain_audio
|
|
535
|
+
|
|
536
|
+
Apply a gain, in decibels, to a region of an audio track - the rest of the
|
|
537
|
+
track passes through unchanged. The region is addressed the same way
|
|
538
|
+
`slice_audio`'s is, in seconds or in video frames, so ducking a scene under
|
|
539
|
+
another (lowering a dialogue track between two timestamps) is one step
|
|
540
|
+
instead of the `slice_audio` → `gain` (a whole-track `normalize_audio` on the
|
|
541
|
+
slice) → `mix_audio` → `rejoin` → `pair_audio` chain that used to be the only
|
|
542
|
+
way to gain part of a track rather than all of it. Unlike `slice_audio`, a
|
|
543
|
+
region reaching past the end of the track is clipped to it rather than
|
|
544
|
+
zero-padded - there is no silence there to gain, only the end of the real
|
|
545
|
+
material:
|
|
546
|
+
|
|
547
|
+
```json
|
|
548
|
+
{
|
|
549
|
+
"task": {
|
|
550
|
+
"command": "gain_audio",
|
|
551
|
+
"arguments": {
|
|
552
|
+
"audio": "./dialogue.wav",
|
|
553
|
+
"gain_db": -12,
|
|
554
|
+
"start_frame": 124,
|
|
555
|
+
"num_frames": 48,
|
|
556
|
+
"fps": 24
|
|
557
|
+
}
|
|
558
|
+
},
|
|
559
|
+
"result": { "content_type": "audio/wav", "sample_rate": 44100 }
|
|
560
|
+
}
|
|
561
|
+
```
|
|
562
|
+
|
|
563
|
+
| Argument | Required | Description |
|
|
564
|
+
| -------- | -------- | ----------- |
|
|
565
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a waveform from a previous step, or an earlier step's video generated with a soundtrack (which brings its sample rate along) |
|
|
566
|
+
| `gain_db` | Yes | Gain to apply within the region, in decibels - negative ducks it, positive boosts it |
|
|
567
|
+
| `start_seconds` / `duration_seconds` | No | The region in seconds; either may be omitted |
|
|
568
|
+
| `start_frame` / `num_frames` / `fps` | No | The region in video frames; `fps` is required if either is given, start and count may be omitted |
|
|
569
|
+
| `sample_rate` | With a waveform | Sample rate of a directly passed waveform (files carry their own) |
|
|
570
|
+
|
|
571
|
+
No region argument is required: with every one of them omitted, the gain
|
|
572
|
+
applies to the whole track (#395) - the same "no region means everything"
|
|
573
|
+
reading `mix_audio`'s gains use. To gain everything from some point on
|
|
574
|
+
instead, give just `start_seconds: 0` and leave `duration_seconds` unset (or
|
|
575
|
+
`start_frame: 0` + `fps` and leave `num_frames` unset), which runs to the
|
|
576
|
+
end of the track without needing to already know how long that is.
|
|
577
|
+
|
|
578
|
+
### crossfade_audio
|
|
579
|
+
|
|
580
|
+
Join audio tracks with an equal-power crossfade. Each seam overlaps the two
|
|
581
|
+
tracks by the fade window:
|
|
582
|
+
|
|
583
|
+
```json
|
|
584
|
+
{
|
|
585
|
+
"task": {
|
|
586
|
+
"command": "crossfade_audio",
|
|
587
|
+
"arguments": {
|
|
588
|
+
"audios": "previous_result:slices",
|
|
589
|
+
"crossfade_ms": 75,
|
|
590
|
+
"sample_rate": 44100
|
|
591
|
+
}
|
|
592
|
+
},
|
|
593
|
+
"result": { "content_type": "audio/wav", "sample_rate": 44100 }
|
|
594
|
+
}
|
|
595
|
+
```
|
|
596
|
+
|
|
597
|
+
### fade_audio
|
|
598
|
+
|
|
599
|
+
Fade a track in from silence and out to it. A slice cut out of the middle of a
|
|
600
|
+
piece ends on whatever was sounding at the cut; a fade turns that into an
|
|
601
|
+
ending. The curve is the equal-power cosine the seam joins use:
|
|
602
|
+
|
|
603
|
+
```json
|
|
604
|
+
{
|
|
605
|
+
"task": {
|
|
606
|
+
"command": "fade_audio",
|
|
607
|
+
"arguments": {
|
|
608
|
+
"audio": "previous_result:soundtrack",
|
|
609
|
+
"fade_in_ms": 500,
|
|
610
|
+
"fade_out_ms": 2500,
|
|
611
|
+
"sample_rate": 44100
|
|
612
|
+
}
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
```
|
|
616
|
+
|
|
617
|
+
| Argument | Required | Description |
|
|
618
|
+
| -------- | -------- | ----------- |
|
|
619
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a waveform from a previous step, or an earlier step's video generated with a soundtrack (which brings its sample rate along) |
|
|
620
|
+
| `fade_in_ms` | No | Length of the fade in, from the head of the track (default: 0) |
|
|
621
|
+
| `fade_out_ms` | No | Length of the fade out, to the tail of the track (default: 0) |
|
|
622
|
+
| `sample_rate` | With a waveform | Sample rate of a directly passed waveform (files carry their own) |
|
|
623
|
+
|
|
624
|
+
**Example:** [audio-trim-fade.json](../workflows/templates/audio-trim-fade.json) — slice a generated track to length, then fade the cut into an ending.
|
|
625
|
+
|
|
626
|
+
### normalize_audio
|
|
627
|
+
|
|
628
|
+
Scale a track so its loudest sample sits at a level. Generated music comes out
|
|
629
|
+
wherever the model happened to land - a quiet take needs lifting before it sits
|
|
630
|
+
under a picture, a hot one needs headroom before the encoder. Only the gain
|
|
631
|
+
changes, so the dynamics survive:
|
|
632
|
+
|
|
633
|
+
```json
|
|
634
|
+
{
|
|
635
|
+
"task": {
|
|
636
|
+
"command": "normalize_audio",
|
|
637
|
+
"arguments": {
|
|
638
|
+
"audio": "previous_result:faded",
|
|
639
|
+
"peak_dbfs": -1.0,
|
|
640
|
+
"sample_rate": 44100
|
|
641
|
+
}
|
|
642
|
+
}
|
|
643
|
+
}
|
|
644
|
+
```
|
|
645
|
+
|
|
646
|
+
| Argument | Required | Description |
|
|
647
|
+
| -------- | -------- | ----------- |
|
|
648
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a waveform from a previous step, or an earlier step's video generated with a soundtrack (which brings its sample rate along) |
|
|
649
|
+
| `peak_dbfs` | No | The level the loudest sample is moved to, in dB below full scale (default: -1.0). 0 is full scale |
|
|
650
|
+
| `sample_rate` | With a waveform | Sample rate of a directly passed waveform (files carry their own) |
|
|
651
|
+
|
|
652
|
+
A silent track is returned unchanged.
|
|
653
|
+
|
|
654
|
+
**Example:** [dissolve-between-shots.json](../workflows/templates/dissolve-between-shots.json)
|
|
655
|
+
|
|
656
|
+
**Headroom and clipping warnings.** Saving audio or a video with a muxed
|
|
657
|
+
soundtrack checks the written level against two thresholds, reported in
|
|
658
|
+
`get_job`'s warnings and readable back afterward as `media.peak_dbfs` from
|
|
659
|
+
`get_gallery_metadata`:
|
|
660
|
+
|
|
661
|
+
- `audio_no_headroom` fires when a plain audio file's waveform, before
|
|
662
|
+
encoding, peaks at or above -0.5 dBFS - encoding can push a level that
|
|
663
|
+
already has no headroom over full scale.
|
|
664
|
+
- `audio_clipped` fires when the file is decoded back *after* writing and
|
|
665
|
+
measures at or above 0.0 dBFS - the ground truth of what a consumer's
|
|
666
|
+
decoder will actually see, since an encoder's own overshoot varies by
|
|
667
|
+
codec and is not reliably predictable from the pre-encode level.
|
|
668
|
+
|
|
669
|
+
A video mux only ever reports the second one: its pre-encode prediction is
|
|
670
|
+
held rather than emitted, because a mux's overshoot is not reliably positive
|
|
671
|
+
the way a plain audio encode's is. That leaves a real gap between the two
|
|
672
|
+
thresholds - a video whose soundtrack decodes back between -0.5 and 0.0 dBFS
|
|
673
|
+
produces no warning at all, because it predicted risk but measured clean.
|
|
674
|
+
That is the file's own measured level, not a threshold bug: read
|
|
675
|
+
`media.peak_dbfs` against -0.5 and 0.0 to judge a specific file rather than
|
|
676
|
+
relying on the warning alone.
|
|
677
|
+
|
|
678
|
+
### mix_audio
|
|
679
|
+
|
|
680
|
+
Layer tracks on top of one another. `crossfade_audio` puts tracks one after
|
|
681
|
+
another; this puts them on top of each other - a score laid under a film's own
|
|
682
|
+
sound, where the music runs unbroken while the world underneath it is replaced
|
|
683
|
+
at every cut:
|
|
684
|
+
|
|
685
|
+
```json
|
|
686
|
+
{
|
|
687
|
+
"task": {
|
|
688
|
+
"command": "mix_audio",
|
|
689
|
+
"arguments": {
|
|
690
|
+
"audios": ["previous_result:soundtrack", "previous_result:world"],
|
|
691
|
+
"gains": [0.5, 1.0],
|
|
692
|
+
"sample_rate": 44100
|
|
693
|
+
}
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
```
|
|
697
|
+
|
|
698
|
+
| Argument | Required | Description |
|
|
699
|
+
| -------- | -------- | ----------- |
|
|
700
|
+
| `audios` | Yes | The tracks to layer - waveforms, audio or video file paths, or videos generated with a soundtrack |
|
|
701
|
+
| `gains` | No | One plain multiplier per track, in the same order - not decibels. Defaults to unity on every track |
|
|
702
|
+
| `sample_rate` | With a raw waveform | Sample rate of the waveforms. Required unless every track brings its own; given here it wins |
|
|
703
|
+
|
|
704
|
+
Tracks of different lengths are padded with silence to the longest, so a score
|
|
705
|
+
shorter than the picture leaves the tail dry rather than cutting the picture
|
|
706
|
+
down to fit. Summing can push peaks past full scale and the sum is *not*
|
|
707
|
+
rescaled - follow it with `normalize_audio` to bring the peak back down.
|
|
708
|
+
|
|
709
|
+
**Example:** [dissolve-between-shots.json](../workflows/templates/dissolve-between-shots.json) — a
|
|
710
|
+
generated score mixed under the shots' own audio.
|
|
711
|
+
|
|
712
|
+
### loop_audio
|
|
713
|
+
|
|
714
|
+
Make a bed of a given length out of a short recording — the room tone laid
|
|
715
|
+
under a whole cut, which is the only complete fix for the hole at a seam. Each
|
|
716
|
+
shot in a cut carries its own room and nothing runs underneath the join;
|
|
717
|
+
a continuous bed does, the way a location's room tone is laid under a dialogue
|
|
718
|
+
scene so the edits stop being audible:
|
|
719
|
+
|
|
720
|
+
```json
|
|
721
|
+
{
|
|
722
|
+
"task": {
|
|
723
|
+
"command": "loop_audio",
|
|
724
|
+
"arguments": {
|
|
725
|
+
"audio": "previous_result:room_tone",
|
|
726
|
+
"target_frames": 620,
|
|
727
|
+
"fps": 24,
|
|
728
|
+
"crossfade_ms": 250
|
|
729
|
+
}
|
|
730
|
+
}
|
|
731
|
+
}
|
|
732
|
+
```
|
|
733
|
+
|
|
734
|
+
| Argument | Required | Description |
|
|
735
|
+
| -------- | -------- | ----------- |
|
|
736
|
+
| `audio` | Yes | Path or URL of an audio or video file, a video generated with a soundtrack (which brings its sample rate along), or a waveform |
|
|
737
|
+
| `duration_seconds` | One of | How long the bed should be, in seconds |
|
|
738
|
+
| `target_frames` / `fps` | One of | How long the bed should be, in video frames — how a bed is matched to a cut exactly |
|
|
739
|
+
| `crossfade_ms` | No | Crossfade at each loop point, clamped to the material available (default 250) |
|
|
740
|
+
| `sample_rate` | With a waveform | Sample rate of a waveform passed directly; given for a file or a video it overrides the rate they carry |
|
|
741
|
+
|
|
742
|
+
Laps are joined with an equal-power crossfade rather than butted together, so
|
|
743
|
+
the loop point itself is not a click. That only smooths the seam: a transient
|
|
744
|
+
in the source (a hit, a swell) still recurs once per lap at full strength, so
|
|
745
|
+
the loop still reads as a level pulse at the lap rate — measured at 9.3 dB on
|
|
746
|
+
a source with one such transient. Pick a source with even internal level to
|
|
747
|
+
avoid the pulse; the crossfade does not remove it. The source is used whole
|
|
748
|
+
every lap and only the last one is trimmed, so the bed lands exactly on the
|
|
749
|
+
requested length; a source longer than the request is trimmed to it.
|
|
750
|
+
|
|
751
|
+
The bed is laid under the cut with `mix_audio` and attached to the picture with
|
|
752
|
+
`pair_audio`:
|
|
753
|
+
|
|
754
|
+
```json
|
|
755
|
+
{ "name": "bed", "task": { "command": "loop_audio",
|
|
756
|
+
"arguments": { "audio": "previous_result:room_tone",
|
|
757
|
+
"target_frames": 620, "fps": 24 } } },
|
|
758
|
+
{ "name": "mixed", "task": { "command": "mix_audio",
|
|
759
|
+
"arguments": { "audios": ["previous_result:episode",
|
|
760
|
+
"previous_result:bed"],
|
|
761
|
+
"gains": [1.0, 0.25] } } },
|
|
762
|
+
{ "name": "cut", "task": { "command": "pair_audio",
|
|
763
|
+
"arguments": { "video": "previous_result:episode",
|
|
764
|
+
"audio": "previous_result:mixed" } },
|
|
765
|
+
"result": { "content_type": "video/mp4", "fps": 24 } }
|
|
766
|
+
```
|
|
767
|
+
|
|
768
|
+
Where the bed itself comes from is the open question: a few seconds of a
|
|
769
|
+
generated shot's own ambience, cut out with `slice_audio` from a stretch with
|
|
770
|
+
nothing tonal in it, is the material that matches — the room the shots were
|
|
771
|
+
generated in.
|
|
772
|
+
|
|
773
|
+
### resample_audio
|
|
774
|
+
|
|
775
|
+
Convert a track to a different sample rate. A pipeline that conditions on audio
|
|
776
|
+
wants it at its own rate (MiniMax H3 at its audio VAE's), and resampling a
|
|
777
|
+
supplied recording once, up front, feeds it what it already wants:
|
|
778
|
+
|
|
779
|
+
```json
|
|
780
|
+
{
|
|
781
|
+
"task": {
|
|
782
|
+
"command": "resample_audio",
|
|
783
|
+
"arguments": {
|
|
784
|
+
"audio": "previous_result:edit",
|
|
785
|
+
"target_sample_rate": 44100
|
|
786
|
+
}
|
|
787
|
+
}
|
|
788
|
+
}
|
|
789
|
+
```
|
|
790
|
+
|
|
791
|
+
| Argument | Required | Description |
|
|
792
|
+
| -------- | -------- | ----------- |
|
|
793
|
+
| `audio` | Yes | Path or URL of an audio or video file, a video generated with a soundtrack (which brings its sample rate along), or a waveform |
|
|
794
|
+
| `target_sample_rate` | Yes | The rate to convert to |
|
|
795
|
+
| `sample_rate` | With a waveform | Sample rate of a waveform passed directly; given for a file or a video it overrides the rate they carry |
|
|
796
|
+
|
|
797
|
+
A track already at the target rate is returned untouched. The conversion is
|
|
798
|
+
PyAV's, which dw already needs for video - no torchaudio dependency.
|
|
799
|
+
|
|
800
|
+
Every audio task returns the waveform *and* the rate it is at, so one chains
|
|
801
|
+
into the next without the rate being restated: a `resample_audio` fed
|
|
802
|
+
`previous_result:` from a `slice_audio` takes the source rate from the slice. A
|
|
803
|
+
`sample_rate` given on the step still wins, and one declared on the step's
|
|
804
|
+
`result` still decides what is written to disk.
|
|
805
|
+
|
|
806
|
+
**Example:** [assemble-and-score.json](../workflows/templates/assemble-and-score.json)
|
|
807
|
+
|
|
808
|
+
### compress_audio
|
|
809
|
+
|
|
810
|
+
Shape a track's dynamics with an envelope-follower - a compressor, a limiter
|
|
811
|
+
and a gate are the same algorithm with different knob settings, so one task
|
|
812
|
+
covers all three through `mode`:
|
|
813
|
+
|
|
814
|
+
```json
|
|
815
|
+
{
|
|
816
|
+
"task": {
|
|
817
|
+
"command": "compress_audio",
|
|
818
|
+
"arguments": {
|
|
819
|
+
"audio": "previous_result:mixed",
|
|
820
|
+
"threshold_dbfs": -18.0,
|
|
821
|
+
"ratio": 4.0,
|
|
822
|
+
"attack_ms": 10,
|
|
823
|
+
"release_ms": 100,
|
|
824
|
+
"sample_rate": 44100
|
|
825
|
+
}
|
|
826
|
+
}
|
|
827
|
+
}
|
|
828
|
+
```
|
|
829
|
+
|
|
830
|
+
| Argument | Required | Description |
|
|
831
|
+
| -------- | -------- | ----------- |
|
|
832
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a waveform from a previous step, or an earlier step's video generated with a soundtrack (which brings its sample rate along) |
|
|
833
|
+
| `threshold_dbfs` | Yes | The level the envelope is measured against, in dB below full scale. Cannot be above 0 |
|
|
834
|
+
| `ratio` | No | How hard the reduction is above the threshold, in `compress`/`gate` mode (default: 4.0). Ignored in `limit` mode, which always holds the signal at the threshold |
|
|
835
|
+
| `attack_ms` | No | How fast the envelope rises to a louder signal (default: 10.0). 0 means instantly |
|
|
836
|
+
| `release_ms` | No | How fast the envelope falls back after a louder signal ends (default: 100.0). 0 means instantly |
|
|
837
|
+
| `mode` | No | `compress` (turn down what's above the threshold), `limit` (hold the signal at the threshold), or `gate` (turn down what's below the threshold) (default: `compress`) |
|
|
838
|
+
| `sample_rate` | With a waveform | Sample rate of a directly passed waveform (files carry their own) |
|
|
839
|
+
|
|
840
|
+
A silent track is returned unchanged.
|
|
841
|
+
|
|
842
|
+
### filter_audio
|
|
843
|
+
|
|
844
|
+
Run a track through a single biquad filter stage - trimming the frequencies a
|
|
845
|
+
mix doesn't need, or carving out room for another element:
|
|
846
|
+
|
|
847
|
+
```json
|
|
848
|
+
{
|
|
849
|
+
"task": {
|
|
850
|
+
"command": "filter_audio",
|
|
851
|
+
"arguments": {
|
|
852
|
+
"audio": "previous_result:world",
|
|
853
|
+
"cutoff_hz": 120,
|
|
854
|
+
"kind": "highpass",
|
|
855
|
+
"sample_rate": 44100
|
|
856
|
+
}
|
|
857
|
+
}
|
|
858
|
+
}
|
|
859
|
+
```
|
|
860
|
+
|
|
861
|
+
| Argument | Required | Description |
|
|
862
|
+
| -------- | -------- | ----------- |
|
|
863
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a waveform from a previous step, or an earlier step's video generated with a soundtrack (which brings its sample rate along) |
|
|
864
|
+
| `cutoff_hz` | Yes | The filter's corner frequency. Must be below the Nyquist frequency (half the sample rate) |
|
|
865
|
+
| `kind` | No | `lowpass`, `highpass`, `bandpass`, or `notch` (default: `lowpass`) |
|
|
866
|
+
| `q` | No | The filter's resonance/bandwidth (default: 0.707, a Butterworth response) |
|
|
867
|
+
| `sample_rate` | With a waveform | Sample rate of a directly passed waveform (files carry their own) |
|
|
868
|
+
|
|
869
|
+
A silent track is returned unchanged.
|
|
870
|
+
|
|
871
|
+
### analyze_audio
|
|
872
|
+
|
|
873
|
+
Measure a track without changing it - peak and RMS level, crest factor, and a
|
|
874
|
+
rough low/mid/high spectral balance, the numbers a `compress_audio` or
|
|
875
|
+
`filter_audio` step downstream is tuned against rather than guessed at:
|
|
876
|
+
|
|
877
|
+
```json
|
|
878
|
+
{
|
|
879
|
+
"task": {
|
|
880
|
+
"command": "analyze_audio",
|
|
881
|
+
"arguments": {
|
|
882
|
+
"audio": "previous_result:mixed",
|
|
883
|
+
"sample_rate": 44100
|
|
884
|
+
}
|
|
885
|
+
}
|
|
886
|
+
}
|
|
887
|
+
```
|
|
888
|
+
|
|
889
|
+
| Argument | Required | Description |
|
|
890
|
+
| -------- | -------- | ----------- |
|
|
891
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a waveform from a previous step, or an earlier step's video generated with a soundtrack (which brings its sample rate along) |
|
|
892
|
+
| `sample_rate` | With a waveform | Sample rate of a directly passed waveform (files carry their own) |
|
|
893
|
+
|
|
894
|
+
Returns a dict, not a track: `peak_dbfs`, `rms_dbfs`, `crest_factor_db`,
|
|
895
|
+
`low_dbfs` (20-250 Hz), `mid_dbfs` (250-4000 Hz), `high_dbfs` (4000-20000 Hz).
|
|
896
|
+
The three bands are each a share of the track's total power on the same
|
|
897
|
+
scale as `rms_dbfs` (their powers sum to it), so the loudest band sits near
|
|
898
|
+
`rms_dbfs` rather than tens of dB under it - comparable to `compress_audio`'s
|
|
899
|
+
`threshold_dbfs`. A silent track, or a band with no content at the track's
|
|
900
|
+
sample rate, reads as `null` rather than `-inf`.
|
|
901
|
+
|
|
902
|
+
## Assessment Probes
|
|
903
|
+
|
|
904
|
+
Three read-only commands measure a finished cut and say where to look -
|
|
905
|
+
`analyze_shots`, `analyze_seams`, `analyze_sync_drift`. Each takes a video
|
|
906
|
+
(a stored file - `asset:`, `output:` or a path, read straight from disk
|
|
907
|
+
rather than decoded first - or the video an earlier step returned; not a
|
|
908
|
+
URL, whose download is a bare frame list with no soundtrack) and answers one JSON
|
|
909
|
+
document: every measurement it took, plus `findings` (the measurements that
|
|
910
|
+
crossed a rule in the table below), `rules_applied` (the rule names the probe
|
|
911
|
+
checked) and `shots_source` (where the shot list came from). A probe reads
|
|
912
|
+
the file streaming - a 64x36 grey thumbnail per frame and the soundtrack,
|
|
913
|
+
never a full frame list - so it runs on a cut of any length.
|
|
914
|
+
|
|
915
|
+
Findings are places to look, not verdicts: nothing in the engine acts on
|
|
916
|
+
one, no run fails for one, and a finding someone has looked at and accepted
|
|
917
|
+
is simply left alone.
|
|
918
|
+
|
|
919
|
+
A probe's `result` must save as JSON:
|
|
920
|
+
|
|
921
|
+
```json
|
|
922
|
+
{
|
|
923
|
+
"task": {
|
|
924
|
+
"command": "analyze_seams",
|
|
925
|
+
"arguments": {
|
|
926
|
+
"video": "output:<identity>/latest/final/cut.mp4"
|
|
927
|
+
}
|
|
928
|
+
},
|
|
929
|
+
"result": { "content_type": "application/json" }
|
|
930
|
+
}
|
|
931
|
+
```
|
|
932
|
+
|
|
933
|
+
Any other `content_type` (or none) fails validation - a JSON document can
|
|
934
|
+
only be saved whole under `application/json`; every other content type
|
|
935
|
+
would explode it key by key or die trying to write a number.
|
|
936
|
+
|
|
937
|
+
Shot boundaries come, in order: the step's own `shots` argument, the shots
|
|
938
|
+
carried by the video an earlier step returned, the run manifest beside the
|
|
939
|
+
file, and otherwise the whole file is treated as one shot. `shots_source`
|
|
940
|
+
reports which - `argument`, `artifact`, `manifest`, or `none`.
|
|
941
|
+
|
|
942
|
+
### analyze_shots
|
|
943
|
+
|
|
944
|
+
Each shot's level and spectral balance, and how far apart the shots sit:
|
|
945
|
+
|
|
946
|
+
| Field | Meaning |
|
|
947
|
+
| ----- | ------- |
|
|
948
|
+
| `shots[].name` | The shot's name |
|
|
949
|
+
| `shots[].start_frame` / `num_frames` | The shot's frame range, as the shot record gave it |
|
|
950
|
+
| `shots[].peak_dbfs` | Peak level within the shot |
|
|
951
|
+
| `shots[].rms_dbfs` | RMS level within the shot |
|
|
952
|
+
| `shots[].crest_db` | `peak_dbfs` minus `rms_dbfs` |
|
|
953
|
+
| `shots[].low_dbfs` / `mid_dbfs` / `high_dbfs` | Spectral balance (20-250 Hz / 250-4000 Hz / 4000-20000 Hz), on the same scale as `rms_dbfs` |
|
|
954
|
+
| `shots[].samples` | Whether the shot's sample span was `recorded` (carried by the shot record) or `derived` (scaled from its frames) |
|
|
955
|
+
| `rms_range_db` | The spread between the loudest and quietest voiced shot |
|
|
956
|
+
| `has_audio` | Whether the file carries a soundtrack at all |
|
|
957
|
+
|
|
958
|
+
### analyze_seams
|
|
959
|
+
|
|
960
|
+
Every seam between shots, audio and picture:
|
|
961
|
+
|
|
962
|
+
| Field | Meaning |
|
|
963
|
+
| ----- | ------- |
|
|
964
|
+
| `seams[].seam` | The seam's index (1-based) |
|
|
965
|
+
| `seams[].between` | `[previous shot name, next shot name]` |
|
|
966
|
+
| `seams[].seconds` | Where the seam sits in the file |
|
|
967
|
+
| `seams[].kind` | `cut` or `dissolve` (a dissolve has `overlap_frames`) |
|
|
968
|
+
| `seams[].hard_cut` | Whether the incoming shot is marked `hard_cut: true` |
|
|
969
|
+
| `seams[].before_shot_rms_dbfs` / `after_shot_rms_dbfs` | RMS level of the whole shot either side of the seam |
|
|
970
|
+
| `seams[].level_step_db` | The absolute difference between those two shot levels. Shot against shot, not the audio at the seam's edges: a take's own tail and head can sit 20 dB apart, which is not a step the cut made |
|
|
971
|
+
| `seams[].before_rms_dbfs` / `after_rms_dbfs` | RMS level of the 0.25 s either side of the seam - what `seam_hole`'s both-sides-voiced guard reads |
|
|
972
|
+
| `seams[].floor_dbfs` | RMS level of the join itself (the fade, or a short window centred on a cut) |
|
|
973
|
+
| `seams[].click_db` | How far a spike at the join peaks above its immediate neighbours |
|
|
974
|
+
| `seams[].spectral_shift` | How much the low/mid/high balance shifts across the seam (0-1) |
|
|
975
|
+
| `seams[].frame_delta` | The largest single-frame picture change across the seam |
|
|
976
|
+
| `seams[].typical_delta` | The larger shot's own typical frame-to-frame change, floored |
|
|
977
|
+
| `seams[].jump_ratio` | `frame_delta` divided by `typical_delta` |
|
|
978
|
+
|
|
979
|
+
### analyze_sync_drift
|
|
980
|
+
|
|
981
|
+
How far the soundtrack sits from the picture, shot by shot and over the
|
|
982
|
+
whole file:
|
|
983
|
+
|
|
984
|
+
| Field | Meaning |
|
|
985
|
+
| ----- | ------- |
|
|
986
|
+
| `shots[].name` | The shot's name |
|
|
987
|
+
| `shots[].start_offset_ms` | How far the audio sits from the picture at the shot's start |
|
|
988
|
+
| `shots[].end_offset_ms` | How far the audio sits from the picture at the shot's end |
|
|
989
|
+
| `max_offset_ms` | The largest `end_offset_ms` across all shots, by magnitude |
|
|
990
|
+
| `video_seconds` / `audio_seconds` | Each stream's own duration |
|
|
991
|
+
| `length_delta_ms` | `audio_seconds` minus `video_seconds` |
|
|
992
|
+
|
|
993
|
+
### Rules
|
|
994
|
+
|
|
995
|
+
Each rule names the probe and field it reads, how the value is compared to
|
|
996
|
+
its threshold, and the severity of a crossing:
|
|
997
|
+
|
|
998
|
+
| Rule | Probe | Field | Threshold | Severity |
|
|
999
|
+
| ---- | ----- | ----- | --------- | -------- |
|
|
1000
|
+
| `shot_level_spread` | `analyze_shots` | `rms_range_db` | >= 6.0 dB | warn |
|
|
1001
|
+
| `seam_level_step` | `analyze_seams` | `level_step_db` | > 3.0 dB | warn |
|
|
1002
|
+
| `seam_click` | `analyze_seams` | `click_db` | > 12.0 dB | warn |
|
|
1003
|
+
| `seam_hole` | `analyze_seams` | `floor_dbfs` | < -50.0 dBFS | warn |
|
|
1004
|
+
| `seam_frame_jump` | `analyze_seams` | `jump_ratio` | > 25.0 | info |
|
|
1005
|
+
| `sync_drift` | `analyze_sync_drift` | `end_offset_ms` | > 40.0 ms (magnitude) | warn |
|
|
1006
|
+
| `sync_length` | `analyze_sync_drift` | `length_delta_ms` | > 40.0 ms (magnitude) | warn |
|
|
1007
|
+
|
|
1008
|
+
Two rules carry a guard beyond the threshold: `seam_hole` only fires while
|
|
1009
|
+
both sides of the seam are voiced above -30 dBFS (a quiet join between two
|
|
1010
|
+
quiet shots is not a hole, it's a pause the shots themselves hold), and
|
|
1011
|
+
`seam_frame_jump` is skipped at a seam whose incoming shot is marked
|
|
1012
|
+
`hard_cut: true` - a cut meant as a cut.
|
|
1013
|
+
|
|
1014
|
+
A `shots` record reaching past the file's own length is a separate finding,
|
|
1015
|
+
`shot_span_overrun`, on all three probes - not a threshold crossing, since
|
|
1016
|
+
the engine clips the record to the file before any of the rules above run.
|
|
1017
|
+
`validate_workflow` reports the same mistake ahead of the run when the
|
|
1018
|
+
video's length is already knowable (a `shots` argument against an
|
|
1019
|
+
`asset:`/literal video); a `previous_result:`/`output:` video not yet
|
|
1020
|
+
written is left to the finding.
|
|
1021
|
+
|
|
1022
|
+
`list_tasks` names the probes in their own `assessment` list, alongside
|
|
1023
|
+
`commands`, so a caller looking for a way to check a cut can find them
|
|
1024
|
+
without reading every command's schema.
|
|
1025
|
+
|
|
1026
|
+
## Data Gathering
|
|
1027
|
+
|
|
1028
|
+
### gather_images
|
|
1029
|
+
|
|
1030
|
+
Load images from URLs and/or file glob patterns:
|
|
1031
|
+
|
|
1032
|
+
```json
|
|
1033
|
+
{
|
|
1034
|
+
"task": {
|
|
1035
|
+
"command": "gather_images",
|
|
1036
|
+
"arguments": {
|
|
1037
|
+
"urls": ["https://example.com/a.jpg", "https://example.com/b.jpg"],
|
|
1038
|
+
"glob": "./images/*.jpg"
|
|
1039
|
+
}
|
|
1040
|
+
}
|
|
1041
|
+
}
|
|
1042
|
+
```
|
|
1043
|
+
|
|
1044
|
+
Returns a list of images that can be referenced by later steps with `previous_result:`.
|
|
1045
|
+
|
|
1046
|
+
### gather_videos
|
|
1047
|
+
|
|
1048
|
+
Same as `gather_images` but for video files. Each video comes back as one
|
|
1049
|
+
artifact holding its frames and whatever audio was muxed alongside them, so a
|
|
1050
|
+
step referencing this one iterates over videos rather than over frames.
|
|
1051
|
+
|
|
1052
|
+
To *join* videos that are already on disk, give their paths to `concat_videos`
|
|
1053
|
+
directly rather than gathering them first: a `previous_result` reference to a
|
|
1054
|
+
gather step fans the consuming step out over the gathered videos instead of
|
|
1055
|
+
handing it all of them at once.
|
|
1056
|
+
|
|
1057
|
+
### gather_inputs
|
|
1058
|
+
|
|
1059
|
+
Pass through arguments directly. Useful for organizing data flow.
|
|
1060
|
+
|
|
1061
|
+
## Videos in image tasks
|
|
1062
|
+
|
|
1063
|
+
Every image command - the upscalers, face restoration, segmentation and the
|
|
1064
|
+
image processors - takes a video where it takes an image: an `AudioVideo` from
|
|
1065
|
+
a generation, `concat_videos` or `dissolve_videos` step, or a frame array from
|
|
1066
|
+
`video_frames`. The command runs over the frames one at a time and returns one
|
|
1067
|
+
video artifact, its soundtrack carried through untouched, so a generated clip
|
|
1068
|
+
can be upscaled without losing what was generated alongside it:
|
|
1069
|
+
|
|
1070
|
+
```json
|
|
1071
|
+
{
|
|
1072
|
+
"task": {
|
|
1073
|
+
"command": "upscale",
|
|
1074
|
+
"arguments": {
|
|
1075
|
+
"image": "previous_result:generate_video",
|
|
1076
|
+
"model_name": "Kim2091/UltraSharp"
|
|
1077
|
+
}
|
|
1078
|
+
},
|
|
1079
|
+
"result": { "content_type": "video/mp4", "fps": 24 }
|
|
1080
|
+
}
|
|
1081
|
+
```
|
|
1082
|
+
|
|
1083
|
+
Captioning (`image_to_text`) is the exception - describe a frame, taken with
|
|
1084
|
+
`get_first_frame`, rather than a video.
|
|
1085
|
+
|
|
1086
|
+
## Image Upscaling
|
|
1087
|
+
|
|
1088
|
+
Upscale images using spandrel-compatible super-resolution models (ESRGAN, SwinIR, HAT, DAT, and 40+ other architectures). Models are auto-detected from weight files.
|
|
1089
|
+
|
|
1090
|
+
```json
|
|
1091
|
+
{
|
|
1092
|
+
"task": {
|
|
1093
|
+
"command": "upscale",
|
|
1094
|
+
"arguments": {
|
|
1095
|
+
"image": "previous_result:generate",
|
|
1096
|
+
"model_name": "Kim2091/UltraSharp",
|
|
1097
|
+
"filename": "4x-UltraSharp.pth"
|
|
1098
|
+
}
|
|
1099
|
+
}
|
|
1100
|
+
}
|
|
1101
|
+
```
|
|
1102
|
+
|
|
1103
|
+
| Argument | Required | Description |
|
|
1104
|
+
| -------- | -------- | ----------- |
|
|
1105
|
+
| `image` | Yes | PIL Image or `previous_result:` reference - a video runs frame by frame, see [Videos in image tasks](#videos-in-image-tasks) |
|
|
1106
|
+
| `model_name` | Yes | HuggingFace repo ID or local file path |
|
|
1107
|
+
| `filename` | No | Specific weight file in a HF repo (auto-detected if only one) |
|
|
1108
|
+
| `tile_size` | No | Tile size for large images (default: 512) |
|
|
1109
|
+
| `tile_overlap` | No | Overlap between tiles in pixels (default: 32) |
|
|
1110
|
+
|
|
1111
|
+
Large images are automatically tiled to avoid GPU memory issues. Models can be loaded from HuggingFace Hub repos or local `.pth`/`.safetensors` files.
|
|
1112
|
+
|
|
1113
|
+
**Examples:**
|
|
1114
|
+
- [upscale-spandrel.json](../workflows/templates/upscale-spandrel.json) — Upscale any existing image 4x.
|
|
1115
|
+
- [upscale-spandrel.json](../workflows/templates/upscale-spandrel.json) — Upscale an image you already have; there is no generation step, so the input is a path or URL.
|
|
1116
|
+
|
|
1117
|
+
## Diffusion Upscaling
|
|
1118
|
+
|
|
1119
|
+
Upscale images using Stable Diffusion upscale pipelines. Text-guided upscaling with better detail recovery than traditional super-resolution, especially for faces and textures.
|
|
1120
|
+
|
|
1121
|
+
Two modes are available:
|
|
1122
|
+
- **x4** (default): `StableDiffusionUpscalePipeline` — 4x upscale via `stabilityai/stable-diffusion-x4-upscaler`
|
|
1123
|
+
- **x2**: `StableDiffusionLatentUpscalePipeline` — 2x upscale via `stabilityai/sd-x2-latent-upscaler`
|
|
1124
|
+
|
|
1125
|
+
```json
|
|
1126
|
+
{
|
|
1127
|
+
"task": {
|
|
1128
|
+
"command": "diffusion_upscale",
|
|
1129
|
+
"arguments": {
|
|
1130
|
+
"image": "previous_result:generate",
|
|
1131
|
+
"prompt": "high quality, detailed",
|
|
1132
|
+
"negative_prompt": "blurry, low quality, artifacts",
|
|
1133
|
+
"mode": "x4"
|
|
1134
|
+
}
|
|
1135
|
+
}
|
|
1136
|
+
}
|
|
1137
|
+
```
|
|
1138
|
+
|
|
1139
|
+
| Argument | Required | Description |
|
|
1140
|
+
| -------- | -------- | ----------- |
|
|
1141
|
+
| `image` | Yes | PIL Image or `previous_result:` reference - a video runs frame by frame, see [Videos in image tasks](#videos-in-image-tasks) |
|
|
1142
|
+
| `prompt` | No | Text guidance for upscaling (default: "") |
|
|
1143
|
+
| `negative_prompt` | No | Negative text guidance (default: none) |
|
|
1144
|
+
| `mode` | No | `"x4"` or `"x2"` (default: `"x4"`) |
|
|
1145
|
+
| `model_name` | No | Override the default model for the selected mode |
|
|
1146
|
+
| `num_inference_steps` | No | Denoising steps (default: 25) |
|
|
1147
|
+
| `guidance_scale` | No | Classifier-free guidance scale (default: 9.0) |
|
|
1148
|
+
| `noise_level` | No | Noise level for x4 mode (default: 20, ignored for x2) |
|
|
1149
|
+
|
|
1150
|
+
**Examples:**
|
|
1151
|
+
- [upscale-diffusion.json](../workflows/templates/upscale-diffusion.json) — Upscale any existing image. `mode` selects which: `x4` (the default) reaches 2048px, `x2` reaches 1024px through the latent upscaler.
|
|
1152
|
+
- [upscale-diffusion.json](../workflows/templates/upscale-diffusion.json) — Prompt-guided upscale of an image you already have, with no generation step.
|
|
1153
|
+
|
|
1154
|
+
## Face Restoration
|
|
1155
|
+
|
|
1156
|
+
Restore and enhance faces in images using spandrel-compatible face restoration models (GFPGAN, CodeFormer, RestoreFormer). Uses facexlib for face detection and alignment, then runs each detected face through the restoration model.
|
|
1157
|
+
|
|
1158
|
+
```json
|
|
1159
|
+
{
|
|
1160
|
+
"task": {
|
|
1161
|
+
"command": "restore_faces",
|
|
1162
|
+
"arguments": {
|
|
1163
|
+
"image": "previous_result:generate",
|
|
1164
|
+
"model_name": "leonelhs/gfpgan",
|
|
1165
|
+
"filename": "GFPGANv1.4.pth"
|
|
1166
|
+
}
|
|
1167
|
+
}
|
|
1168
|
+
}
|
|
1169
|
+
```
|
|
1170
|
+
|
|
1171
|
+
| Argument | Required | Description |
|
|
1172
|
+
| -------- | -------- | ----------- |
|
|
1173
|
+
| `image` | Yes | PIL Image or `previous_result:` reference - a video runs frame by frame, see [Videos in image tasks](#videos-in-image-tasks) |
|
|
1174
|
+
| `model_name` | Yes | HuggingFace repo ID or local file path |
|
|
1175
|
+
| `filename` | No | Specific weight file in a HF repo (auto-detected if only one) |
|
|
1176
|
+
| `upscale_factor` | No | Background upscale factor (default: 1, no upscaling) |
|
|
1177
|
+
| `face_size` | No | Cropped face size in pixels (default: 512) |
|
|
1178
|
+
| `use_parse` | No | Use face parsing for better blending (default: true) |
|
|
1179
|
+
| `only_center_face` | No | Only restore the largest/center face (default: false) |
|
|
1180
|
+
| `detection_resize` | No | Resize shorter side for detection speed (default: 640) |
|
|
1181
|
+
| `eye_dist_threshold` | No | Skip faces with eye distance below this (default: 5) |
|
|
1182
|
+
| `upsample_img` | No | Pre-upscaled background image (e.g., from a prior upscale step) |
|
|
1183
|
+
|
|
1184
|
+
Models are loaded via spandrel, so any `.pth`/`.safetensors` face restoration weights work. CodeFormer requires `pip install spandrel-extra-arches` (non-commercial license).
|
|
1185
|
+
|
|
1186
|
+
**Example:** [restore-faces.json](../workflows/templates/restore-faces.json) — Generate a portrait, then restore faces with GFPGAN v1.4.
|
|
1187
|
+
|
|
1188
|
+
### Combining with Upscaling
|
|
1189
|
+
|
|
1190
|
+
You can chain upscaling and face restoration. Generate first, upscale the background, then paste restored faces onto the upscaled image:
|
|
1191
|
+
|
|
1192
|
+
```json
|
|
1193
|
+
{
|
|
1194
|
+
"steps": [
|
|
1195
|
+
{
|
|
1196
|
+
"name": "generate",
|
|
1197
|
+
"pipeline": { "..." : "..." },
|
|
1198
|
+
"result": { "content_type": "image/jpeg" }
|
|
1199
|
+
},
|
|
1200
|
+
{
|
|
1201
|
+
"name": "upscale",
|
|
1202
|
+
"task": {
|
|
1203
|
+
"command": "upscale",
|
|
1204
|
+
"arguments": {
|
|
1205
|
+
"image": "previous_result:generate",
|
|
1206
|
+
"model_name": "Kim2091/UltraSharp",
|
|
1207
|
+
"filename": "4x-UltraSharp.pth"
|
|
1208
|
+
}
|
|
1209
|
+
},
|
|
1210
|
+
"result": { "content_type": "image/jpeg" }
|
|
1211
|
+
},
|
|
1212
|
+
{
|
|
1213
|
+
"name": "restore",
|
|
1214
|
+
"task": {
|
|
1215
|
+
"command": "restore_faces",
|
|
1216
|
+
"arguments": {
|
|
1217
|
+
"image": "previous_result:generate",
|
|
1218
|
+
"model_name": "leonelhs/gfpgan",
|
|
1219
|
+
"filename": "GFPGANv1.4.pth",
|
|
1220
|
+
"upscale_factor": 4,
|
|
1221
|
+
"upsample_img": "previous_result:upscale"
|
|
1222
|
+
}
|
|
1223
|
+
},
|
|
1224
|
+
"result": { "content_type": "image/jpeg" }
|
|
1225
|
+
}
|
|
1226
|
+
]
|
|
1227
|
+
}
|
|
1228
|
+
```
|
|
1229
|
+
|
|
1230
|
+
This gives the best results: the super-resolution model handles background detail while the face model handles facial features, composited together at the upscaled resolution.
|
|
1231
|
+
|
|
1232
|
+
## Object Segmentation
|
|
1233
|
+
|
|
1234
|
+
Detect and segment objects using text prompts via GroundingDINO + SAM2. Returns a binary mask image suitable for inpainting workflows.
|
|
1235
|
+
|
|
1236
|
+
```json
|
|
1237
|
+
{
|
|
1238
|
+
"task": {
|
|
1239
|
+
"command": "segment",
|
|
1240
|
+
"arguments": {
|
|
1241
|
+
"image": "previous_result:input_image",
|
|
1242
|
+
"prompt": "dog"
|
|
1243
|
+
}
|
|
1244
|
+
},
|
|
1245
|
+
"result": { "content_type": "image/png" }
|
|
1246
|
+
}
|
|
1247
|
+
```
|
|
1248
|
+
|
|
1249
|
+
| Argument | Required | Description |
|
|
1250
|
+
| -------- | -------- | ----------- |
|
|
1251
|
+
| `image` | Yes | PIL Image or `previous_result:` reference - a video runs frame by frame, see [Videos in image tasks](#videos-in-image-tasks) |
|
|
1252
|
+
| `prompt` | Yes | Text description of object(s) to detect (e.g., "dog", "red car") |
|
|
1253
|
+
| `model_name` | No | GroundingDINO model ID (default: `IDEA-Research/grounding-dino-base`) |
|
|
1254
|
+
| `sam_model_name` | No | SAM2 model ID (default: `facebook/sam2-hiera-large`) |
|
|
1255
|
+
| `threshold` | No | Detection confidence threshold (default: 0.3) |
|
|
1256
|
+
| `invert` | No | Invert the output mask (default: false) |
|
|
1257
|
+
|
|
1258
|
+
Returns a grayscale PIL Image (mode "L") — white (255) for detected objects, black (0) for background. Use with inpainting pipelines like FluxFillPipeline.
|
|
1259
|
+
|
|
1260
|
+
**Examples:**
|
|
1261
|
+
|
|
1262
|
+
- [segment.json](../workflows/templates/segment.json) — Segment an object from an image
|
|
1263
|
+
- [segment-and-inpaint.json](../workflows/templates/segment-and-inpaint.json) — Segment, then inpaint the masked region
|
|
1264
|
+
|
|
1265
|
+
## Image Captioning
|
|
1266
|
+
|
|
1267
|
+
Generate text captions from images using a vision-language model.
|
|
1268
|
+
|
|
1269
|
+
Transformers 5 removed the dedicated `image-to-text` pipeline this task used to build, along with the BLIP/ViT-GPT2/GIT captioning models that ran on it. Captioning now goes through the same `image-text-to-text` pipeline as any other VLM, so `model_name` needs a vision-language model (SmolVLM, Qwen2.5-VL, LLaVA, etc.) and `prompt` is a question put to the model rather than a text fragment to continue.
|
|
1270
|
+
|
|
1271
|
+
```json
|
|
1272
|
+
{
|
|
1273
|
+
"task": {
|
|
1274
|
+
"command": "image_to_text",
|
|
1275
|
+
"arguments": {
|
|
1276
|
+
"image": "previous_result:input_image"
|
|
1277
|
+
}
|
|
1278
|
+
},
|
|
1279
|
+
"result": { "content_type": "text/plain" }
|
|
1280
|
+
}
|
|
1281
|
+
```
|
|
1282
|
+
|
|
1283
|
+
| Argument | Required | Description |
|
|
1284
|
+
| -------- | -------- | ----------- |
|
|
1285
|
+
| `image` | Yes | PIL Image, URL/path, or `previous_result:` reference |
|
|
1286
|
+
| `model_name` | No | HuggingFace vision-language model ID (default: `HuggingFaceTB/SmolVLM-256M-Instruct`) |
|
|
1287
|
+
| `prompt` | No | What to ask about the image (default: `Describe this image.`) — ask a narrower question for a narrower caption |
|
|
1288
|
+
| `system_prompt` | No | System instruction for the model |
|
|
1289
|
+
| `max_new_tokens` | No | Maximum tokens to generate (default: 50) |
|
|
1290
|
+
|
|
1291
|
+
The default model is deliberately tiny, matching the footprint of the old captioning default; it produces short, plain captions. Point `model_name` at something larger for detail.
|
|
1292
|
+
|
|
1293
|
+
Returns a caption string. Save as `text/plain` for `.txt` output, or pass to a downstream step via `previous_result:` as a prompt for image generation.
|
|
1294
|
+
|
|
1295
|
+
For a detailed caption, hand the image to [`text_generation`](#text-generation) with a question and a larger vision-language model; that is what [describe-and-regenerate.json](../workflows/templates/describe-and-regenerate.json) does ahead of its prompt expansion.
|
|
1296
|
+
|
|
1297
|
+
**Examples:**
|
|
1298
|
+
|
|
1299
|
+
- [image-to-text.json](../workflows/templates/image-to-text.json) — Basic captioning with the default model, saves as `.txt`
|
|
1300
|
+
- [image-to-text.json](../workflows/templates/image-to-text.json) — Larger VLM answering a specific question
|
|
1301
|
+
- [describe-and-regenerate.json](../workflows/templates/describe-and-regenerate.json) — Describe an image, expand the caption, then regenerate it
|
|
1302
|
+
|
|
1303
|
+
## Composing Text
|
|
1304
|
+
|
|
1305
|
+
Assemble one block of text out of parts written once. A multi-shot workflow
|
|
1306
|
+
says the same things about its characters in every shot — who they are, what
|
|
1307
|
+
they are wearing, what their voice sounds like — and the engine deliberately
|
|
1308
|
+
has no string interpolation to splice them in with (see the no-interpolation
|
|
1309
|
+
rule in [the workflow guide](WORKFLOW_GUIDE.md)). Composition is the way
|
|
1310
|
+
round it: a part is a *whole* value, and `compose_text` joins parts in order.
|
|
1311
|
+
|
|
1312
|
+
```json
|
|
1313
|
+
{
|
|
1314
|
+
"name": "shot_1_prompt",
|
|
1315
|
+
"task": {
|
|
1316
|
+
"command": "compose_text",
|
|
1317
|
+
"arguments": {
|
|
1318
|
+
"parts": [
|
|
1319
|
+
"variable:character_a_bible",
|
|
1320
|
+
"variable:character_a_voice",
|
|
1321
|
+
"variable:shot_1_action"
|
|
1322
|
+
],
|
|
1323
|
+
"separator": "\n\n"
|
|
1324
|
+
}
|
|
1325
|
+
}
|
|
1326
|
+
},
|
|
1327
|
+
{
|
|
1328
|
+
"name": "shot_1",
|
|
1329
|
+
"pipeline": { "arguments": { "prompt": "previous_result:shot_1_prompt" } }
|
|
1330
|
+
}
|
|
1331
|
+
```
|
|
1332
|
+
|
|
1333
|
+
| Argument | Required | Description |
|
|
1334
|
+
| -------- | -------- | ----------- |
|
|
1335
|
+
| `parts` | Yes | The parts to join, in order — each a whole value, usually a `variable:`, `prompt:` or `previous_result:` reference. Numbers are written out; `null` is dropped, so an optional part can be a variable left null |
|
|
1336
|
+
| `separator` | No | What goes between the parts (default: a blank line, the paragraph break the prompt formats use) |
|
|
1337
|
+
| `skip_empty` | No | Drop parts that are null or blank (default `true`). With it off, an empty part still contributes its separator |
|
|
1338
|
+
|
|
1339
|
+
A part that is neither text nor a number is an error, not a coercion: it means
|
|
1340
|
+
the reference in that position resolved to something other than the text meant.
|
|
1341
|
+
|
|
1342
|
+
The parts are positional. A named form (`"{bible} says {line}"`) would be the
|
|
1343
|
+
interpolation the engine does not have, one layer down — so a character bible
|
|
1344
|
+
is a variable named by every shot that needs it, and a voice string written
|
|
1345
|
+
once is checked by being the same value rather than by being compared.
|
|
1346
|
+
|
|
1347
|
+
## Extracting Sections
|
|
1348
|
+
|
|
1349
|
+
Reduce generated text to a known set of labelled sections, dropping anything else:
|
|
1350
|
+
|
|
1351
|
+
```json
|
|
1352
|
+
{
|
|
1353
|
+
"task": {
|
|
1354
|
+
"command": "extract_sections",
|
|
1355
|
+
"arguments": {
|
|
1356
|
+
"text": "previous_result:expand",
|
|
1357
|
+
"sections": ["integrated_multimodal_description", "overall_soundscape", "non_diegetic_music"]
|
|
1358
|
+
}
|
|
1359
|
+
},
|
|
1360
|
+
"result": { "content_type": "text/plain" }
|
|
1361
|
+
}
|
|
1362
|
+
```
|
|
1363
|
+
|
|
1364
|
+
| Argument | Required | Description |
|
|
1365
|
+
| -------- | -------- | ----------- |
|
|
1366
|
+
| `text` | Yes | The generated text, usually a `previous_result:` reference |
|
|
1367
|
+
| `sections` | Yes | Section labels to keep, in the order they should appear |
|
|
1368
|
+
| `keep_preamble` | No | Keep any text before the first label (default: `true`) |
|
|
1369
|
+
|
|
1370
|
+
A section runs from its `label:` to the end of that paragraph, so a blank line ends one and a single newline does not — a field holding one line per item stays intact. Repeats are dropped, missing sections are skipped, and text with no recognised label is returned unchanged.
|
|
1371
|
+
|
|
1372
|
+
This exists because a model asked for a rigid format usually produces it and then keeps going — restating the description, appending a summary, or looping until it runs out of tokens. Prompting against that is unreliable, and at small model sizes adding rules to an already long specification can make adherence worse. Trailing text is not free either: a prompt is conditioning, and a pipeline that does not truncate spends memory and attention on whatever arrives. Keeping the fields that were asked for is deterministic where prompting is not.
|
|
1373
|
+
|
|
1374
|
+
The built-in `h3_context_ir` workflow applies this to its own output, so a workflow delegating to it receives only the fields MiniMax H3 expects.
|
|
1375
|
+
|
|
1376
|
+
## Text Generation / Prompt Expansion
|
|
1377
|
+
|
|
1378
|
+
Generate or expand text using a local language model. Useful for expanding short prompts into detailed image generation prompts, rewriting text, or other text-to-text tasks.
|
|
1379
|
+
|
|
1380
|
+
```json
|
|
1381
|
+
{
|
|
1382
|
+
"task": {
|
|
1383
|
+
"command": "text_generation",
|
|
1384
|
+
"arguments": {
|
|
1385
|
+
"prompt": "a cat on a windowsill",
|
|
1386
|
+
"system_prompt": "You are a helpful AI assistant that creates detailed prompts for text to image generative AI. When supplied input generate only the prompt, no other text."
|
|
1387
|
+
}
|
|
1388
|
+
},
|
|
1389
|
+
"result": { "content_type": "text/plain" }
|
|
1390
|
+
}
|
|
1391
|
+
```
|
|
1392
|
+
|
|
1393
|
+
| Argument | Required | Description |
|
|
1394
|
+
| -------- | -------- | ----------- |
|
|
1395
|
+
| `prompt` | Yes | The user message or short prompt to expand/transform |
|
|
1396
|
+
| `system_prompt` | No | System instruction for the model (e.g., "expand this into a detailed image prompt") |
|
|
1397
|
+
| `model_name` | No | HuggingFace model ID (default: `Qwen/Qwen2.5-1.5B-Instruct`, or `HuggingFaceTB/SmolVLM-256M-Instruct` when an image is supplied) |
|
|
1398
|
+
| `image` | No | PIL Image, URL/path, or `previous_result:` reference — see below |
|
|
1399
|
+
| `repetition_penalty` | No | Vision path only (default: 1.15) — see below |
|
|
1400
|
+
| `generate_kwargs` | No | Anything else to pass to the model's `generate()` — `no_repeat_ngram_size`, `top_p`, `min_new_tokens`. Merged last, so it overrides the settings above |
|
|
1401
|
+
| `max_new_tokens` | No | Maximum tokens to generate (default: 500) |
|
|
1402
|
+
|
|
1403
|
+
### Writing a prompt from a picture
|
|
1404
|
+
|
|
1405
|
+
Supplying `image` switches the task to a vision-language model, so the generated text describes what is actually in the picture instead of what the prompt guesses is there. `model_name` must then name a VLM — a text-only model cannot be loaded as one.
|
|
1406
|
+
|
|
1407
|
+
```json
|
|
1408
|
+
{
|
|
1409
|
+
"task": {
|
|
1410
|
+
"command": "text_generation",
|
|
1411
|
+
"arguments": {
|
|
1412
|
+
"prompt": "Write a video prompt that starts from this picture.",
|
|
1413
|
+
"image": "previous_result:input_image",
|
|
1414
|
+
"model_name": "Qwen/Qwen3-VL-4B-Instruct"
|
|
1415
|
+
}
|
|
1416
|
+
},
|
|
1417
|
+
"result": { "content_type": "text/plain" }
|
|
1418
|
+
}
|
|
1419
|
+
```
|
|
1420
|
+
|
|
1421
|
+
This matters most ahead of an image-conditioned generation step. Those pipelines pin the supplied picture as the first frame, so a prompt written without seeing it will describe a scene the keyframe contradicts and the two conditionings pull against each other. Pass the same image to both and the prompt agrees with the frame it opens on.
|
|
1422
|
+
|
|
1423
|
+
A vision model is large enough to be worth releasing before the generation model loads — see `release_models` in the workflow guide.
|
|
1424
|
+
|
|
1425
|
+
Generation stays greedy so a workflow reproduces, but greedy decoding against a long, rigid format specification makes these models loop — emitting a complete answer and then repeating its closing sections until the token budget runs out. The vision path applies a `repetition_penalty` of 1.15 to stop that. Measured on Qwen3-VL against the MiniMax H3 prompt spec, 1.05 still looped through the whole budget while 1.15 ended on its own at a length matching the format's own guidance. Raise it if a model still repeats itself, or set `1.0` to disable.
|
|
1426
|
+
|
|
1427
|
+
A penalty reins the looping in but does not guarantee the model stops where the format ends; for that, trim the output with `extract_sections` below.
|
|
1428
|
+
|
|
1429
|
+
There is a limit to what a small model will follow. Against the MiniMax H3 spec, neither Qwen3-VL-4B nor 8B produces the `<d>[Language]...</d>` dialogue tag or the `(S1)` speaker ids, whether the idea implies speech or supplies the line verbatim; the 8B is worse on layout, capitalising its section labels. Showing a complete worked example does produce them - by copying the example word for word, which is useless - and a placeholder skeleton does not produce them at all. The visual description these models write is grounded and usable; the dialogue markup is not. Write prompts by hand where a subject has to speak.
|
|
1430
|
+
|
|
1431
|
+
For anything the arguments above do not cover, `generate_kwargs` goes straight to `generate()`:
|
|
1432
|
+
|
|
1433
|
+
```json
|
|
1434
|
+
"arguments": {
|
|
1435
|
+
"prompt": "a cat on a windowsill",
|
|
1436
|
+
"generate_kwargs": { "no_repeat_ngram_size": 25 }
|
|
1437
|
+
}
|
|
1438
|
+
```
|
|
1439
|
+
|
|
1440
|
+
It is merged after everything else, so it can override `repetition_penalty` and the sampling settings as well as add to them.
|
|
1441
|
+
|
|
1442
|
+
**Examples:**
|
|
1443
|
+
|
|
1444
|
+
- [expand-prompt.json](../workflows/templates/expand-prompt.json) — Expand a short prompt and save as `.txt`
|
|
1445
|
+
- [expand-prompt.json](../workflows/templates/expand-prompt.json) — Expand prompt, then generate with Flux
|
|
1446
|
+
|
|
1447
|
+
## Speech Generation
|
|
1448
|
+
|
|
1449
|
+
Speak a line of text with a local text-to-speech model. The result is a waveform carrying the rate its model generated at, so it composes with [`slice_audio`](#slice_audio), [`fade_audio`](#fade_audio) and [`pair_audio`](#pair_audio) directly ([`concat_videos`](#concat_videos) and `dissolve_videos` join videos — pair the track onto a video first).
|
|
1450
|
+
|
|
1451
|
+
```json
|
|
1452
|
+
{
|
|
1453
|
+
"task": {
|
|
1454
|
+
"command": "generate_speech",
|
|
1455
|
+
"arguments": {
|
|
1456
|
+
"text": "The way ahead is longer still.",
|
|
1457
|
+
"voice_preset": "v2/en_speaker_6"
|
|
1458
|
+
}
|
|
1459
|
+
},
|
|
1460
|
+
"result": { "content_type": "audio/wav" }
|
|
1461
|
+
}
|
|
1462
|
+
```
|
|
1463
|
+
|
|
1464
|
+
| Argument | Required | Description |
|
|
1465
|
+
| -------- | -------- | ----------- |
|
|
1466
|
+
| `text` | One of `text`/`messages` | The line to speak |
|
|
1467
|
+
| `messages` | One of `text`/`messages` | Chat-templated input for a model such as VibeVoice that takes a conversation rather than a bare string — a list of `{"role": ..., "content": ...}` dicts, passed straight through as the pipeline's `text_inputs` so the model's own chat template applies. A model with no chat template configured (Bark and friends) raises when handed this instead of `text` |
|
|
1468
|
+
| `model_name` | No | HuggingFace model ID (default: `suno/bark-small`) |
|
|
1469
|
+
| `voice_preset` | No | The speaker, for a model with presets — `v2/en_speaker_0` through `v2/en_speaker_9` for Bark. A model with no processor (a single-voice model such as `facebook/mms-tts-eng`) refuses a `voice_preset` with an error rather than ignoring it |
|
|
1470
|
+
| `speaker_embedding` | No | A reference audio file (typically an `asset:` reference) whose voice a SpeechT5 model should speak in. Reduced to an x-vector with speechbrain's `spkrec-xvect-voxceleb` and injected into `forward_params` as `speaker_embeddings`. A model that isn't SpeechT5 refuses it the same way a single-voice model refuses `voice_preset` |
|
|
1471
|
+
| `forward_params` | No | Passed to the model's forward/generate call |
|
|
1472
|
+
| `generate_kwargs` | No | Ad-hoc generation settings for a generative model — `temperature`, `do_sample` |
|
|
1473
|
+
|
|
1474
|
+
The command's own default is Bark because its voice presets give distinct speakers, which is what two characters in a scene need; `facebook/mms-tts-eng` is a quarter the size and a good override where one voice will do. [`generate-speech.json`](../workflows/templates/generate-speech.json) ships with `facebook/mms-tts-eng` as its own default instead, since a template is usually a single voice. `voice_preset` is a preprocessing argument — it selects the speaker before generation rather than parameterizing it — so naming it here is what makes it reach the processor. Passed through `forward_params` it would be dropped and every character would sound the same.
|
|
1475
|
+
|
|
1476
|
+
`speaker_embedding` is the same kind of preprocessing argument for a SpeechT5 model (`microsoft/speecht5_tts`), which conditions its voice on an x-vector rather than a preset name. A VITS model's speaker is different again — a plain `speaker_id` int, passed through `forward_params` unchanged, since it was never a preprocessing argument and needs no argument of its own here.
|
|
1477
|
+
|
|
1478
|
+
The result needs no `sample_rate`. A generated track carries the rate its model produced it at, and that beats the 44100 default; declaring one still wins over both, for a track whose rate was reported wrong. Every TTS model runs at a different rate, so a declared rate that does not match plays the speech at the wrong speed and pitch without ever failing.
|
|
1479
|
+
|
|
1480
|
+
### Generating a voice to condition on
|
|
1481
|
+
|
|
1482
|
+
The role this earns its place in is voice *timbre reference*, not the track a mouth follows. MiniMax H3 lip-syncs well when it generates the speech itself and poorly when it must follow supplied audio, so its `MiniMaxH3AudioReference` takes a few seconds of a voice to fix timbre, pitch and delivery while H3 still generates the line. Build the reference with `from_previous_result` and the clip's own sample rate comes across with it:
|
|
1483
|
+
|
|
1484
|
+
```json
|
|
1485
|
+
"references": [
|
|
1486
|
+
{
|
|
1487
|
+
"reference_type": "diffusers.modular_pipelines.minimax_h3.MiniMaxH3AudioReference",
|
|
1488
|
+
"from_previous_result": "voice"
|
|
1489
|
+
}
|
|
1490
|
+
]
|
|
1491
|
+
```
|
|
1492
|
+
|
|
1493
|
+
Referencing the same preset in every shot of a scene makes a character's voice a conditioning signal rather than a prose description that has to land identically a dozen times. The other honest uses are a voice that must be matched — a specific delivery H3 will not produce from description alone — and narration over shots where nothing has to lip-sync to it, muxed with [`pair_audio`](#pair_audio).
|
|
1494
|
+
|
|
1495
|
+
A speech model is worth releasing before a video model loads — set `release_models` on the step, as in the example below.
|
|
1496
|
+
|
|
1497
|
+
**Examples:**
|
|
1498
|
+
|
|
1499
|
+
- [generate-speech.json](../workflows/templates/generate-speech.json) — Speak a line and save it as a `.wav`
|
|
1500
|
+
- [voice-timbre-reference.json](../workflows/templates/minimax/voice-timbre-reference.json) — Generate a voice, then condition H3's `<Audio 1>` on it
|
|
1501
|
+
|
|
1502
|
+
## Speech Transcription
|
|
1503
|
+
|
|
1504
|
+
Transcribe spoken audio to text with a local Whisper-class model. The word-correctness of a TTS deliverable — a dropped line, a mid-sentence truncation — can only be inferred from duration and timing arithmetic without this; `transcribe_audio` checks it directly against the text the deliverable was supposed to speak.
|
|
1505
|
+
|
|
1506
|
+
```json
|
|
1507
|
+
{
|
|
1508
|
+
"task": {
|
|
1509
|
+
"command": "transcribe_audio",
|
|
1510
|
+
"arguments": {
|
|
1511
|
+
"audio": "previous_result:speak"
|
|
1512
|
+
}
|
|
1513
|
+
},
|
|
1514
|
+
"result": { "content_type": "text/plain" }
|
|
1515
|
+
}
|
|
1516
|
+
```
|
|
1517
|
+
|
|
1518
|
+
| Argument | Required | Description |
|
|
1519
|
+
| -------- | -------- | ----------- |
|
|
1520
|
+
| `audio` | Yes | Path or URL of an audio file (or of a video file, whose soundtrack is taken), a video with a soundtrack, or a waveform — usually a `previous_result:` reference |
|
|
1521
|
+
| `sample_rate` | No | Sample rate of a waveform passed directly |
|
|
1522
|
+
| `model_name` | No | HuggingFace model ID of a Whisper-class ASR model (default: `openai/whisper-base`) |
|
|
1523
|
+
|
|
1524
|
+
Multi-channel audio is downmixed to mono and resampled to 16 kHz before transcription, since that is what a Whisper-class model is trained on; the source audio itself is untouched. The result is plain text, read with MCP's `get_output_text`.
|
|
1525
|
+
|
|
1526
|
+
**Example:** [transcribe-audio.json](../workflows/templates/transcribe-audio.json) — Transcribe an audio file to text.
|
|
1527
|
+
|
|
1528
|
+
## Frame Interpolation
|
|
1529
|
+
|
|
1530
|
+
Increase video frame rate using RIFE (Real-Time Intermediate Flow Estimation). Takes a video and inserts intermediate frames between each pair. The result is one video artifact without a soundtrack - the frame count changed, so [`pair_audio`](#pair_audio) is how the original track comes back. [interpolate-frames.json](../workflows/templates/interpolate-frames.json) shows the interpolation itself.
|
|
1531
|
+
|
|
1532
|
+
```json
|
|
1533
|
+
{
|
|
1534
|
+
"task": {
|
|
1535
|
+
"command": "interpolate_frames",
|
|
1536
|
+
"arguments": {
|
|
1537
|
+
"video": "previous_result:generate_video",
|
|
1538
|
+
"multiplier": 2
|
|
1539
|
+
}
|
|
1540
|
+
},
|
|
1541
|
+
"result": { "content_type": "video/mp4", "fps": 60 }
|
|
1542
|
+
}
|
|
1543
|
+
```
|
|
1544
|
+
|
|
1545
|
+
| Argument | Required | Description |
|
|
1546
|
+
| -------- | -------- | ----------- |
|
|
1547
|
+
| `video` | Yes | The frames - a frame list, a frame array, or an audio+video pair from a concat or dissolve step (its audio is dropped) - usually a `previous_result:` reference |
|
|
1548
|
+
| `multiplier` | No | Frame count multiplier: 2, 4, or 8 (default: 2) |
|
|
1549
|
+
| `model_name` | No | HuggingFace repo with RIFE v4.13 weights (default: `imaginairy/rife-interpolation`) |
|
|
1550
|
+
| `filename` | No | Weights filename within the repo (default: `rife-flownet-4.13.2.safetensors`) |
|
|
1551
|
+
|
|
1552
|
+
Uses vendored IFNet v4.13 architecture. Weights are downloaded from HuggingFace Hub on first use.
|
|
1553
|
+
|
|
1554
|
+
**Example:** [interpolate-frames.json](../workflows/templates/interpolate-frames.json) — Generate video with Mochi, then 2x interpolate from 30fps to 60fps.
|
|
1555
|
+
|
|
1556
|
+
## Metadata Embedding
|
|
1557
|
+
|
|
1558
|
+
Embed generation parameters in saved images. Enable by setting `embed_metadata: true` in a step's result configuration:
|
|
1559
|
+
|
|
1560
|
+
```json
|
|
1561
|
+
{
|
|
1562
|
+
"result": {
|
|
1563
|
+
"content_type": "image/png",
|
|
1564
|
+
"embed_metadata": true
|
|
1565
|
+
}
|
|
1566
|
+
}
|
|
1567
|
+
```
|
|
1568
|
+
|
|
1569
|
+
| Format | Storage | Notes |
|
|
1570
|
+
| ------ | ------- | ----- |
|
|
1571
|
+
| PNG | Text chunk (`parameters` key) | Always available |
|
|
1572
|
+
| JPEG/WebP | EXIF UserComment | Requires `pip install piexif` |
|
|
1573
|
+
|
|
1574
|
+
Metadata includes step name, model name, and generation arguments (prompt, steps, guidance scale, etc.) as JSON.
|
|
1575
|
+
|
|
1576
|
+
**Example:** [embed-metadata.json](../workflows/templates/embed-metadata.json) — Generate with Flux and embed parameters in PNG.
|
|
1577
|
+
|
|
1578
|
+
## QR Code Generation
|
|
1579
|
+
|
|
1580
|
+
```json
|
|
1581
|
+
{
|
|
1582
|
+
"task": {
|
|
1583
|
+
"command": "qr_code",
|
|
1584
|
+
"arguments": {
|
|
1585
|
+
"qr_code_contents": "https://example.com"
|
|
1586
|
+
}
|
|
1587
|
+
}
|
|
1588
|
+
}
|
|
1589
|
+
```
|
|
1590
|
+
|
|
1591
|
+
| Argument | Required | Description |
|
|
1592
|
+
| -------- | -------- | ----------- |
|
|
1593
|
+
| `qr_code_contents` | Yes | Data to encode (URL, text, etc.) |
|
|
1594
|
+
| `height` | No | Used with `width` to derive output resolution (default: 768) |
|
|
1595
|
+
| `width` | No | Used with `height` to derive output resolution (default: 768) |
|
|
1596
|
+
|
|
1597
|
+
The QR code is generated then resampled to `max(height, width)`, aligned to the nearest 64px multiple.
|
|
1598
|
+
|
|
1599
|
+
**Example:** [qr-code.json](../workflows/templates/qr-code.json) — QR code with artistic ControlNet
|
|
1600
|
+
|
|
1601
|
+
## Chat/Dict Plumbing
|
|
1602
|
+
|
|
1603
|
+
These small tasks glue together multi-step pipelines that mix raw `transformers` components with task steps — for the cases `text_generation` does not cover.
|
|
1604
|
+
|
|
1605
|
+
### format_chat_message
|
|
1606
|
+
|
|
1607
|
+
Build a `text_inputs` chat message list from a system and user message, in the shape a `transformers.pipeline` text-generation call expects:
|
|
1608
|
+
|
|
1609
|
+
```json
|
|
1610
|
+
{
|
|
1611
|
+
"task": {
|
|
1612
|
+
"command": "format_chat_message",
|
|
1613
|
+
"arguments": {
|
|
1614
|
+
"system_prompt": "You are a helpful assistant.",
|
|
1615
|
+
"user_message": "variable:prompt"
|
|
1616
|
+
}
|
|
1617
|
+
}
|
|
1618
|
+
}
|
|
1619
|
+
```
|
|
1620
|
+
|
|
1621
|
+
| Argument | Required | Description |
|
|
1622
|
+
| -------- | -------- | ----------- |
|
|
1623
|
+
| `system_prompt` | Yes | System instruction |
|
|
1624
|
+
| `user_message` | Yes | User message content |
|
|
1625
|
+
|
|
1626
|
+
Returns `{"text_inputs": [{"role": "system", ...}, {"role": "user", ...}]}`. Pass the result to a `transformers.pipeline` step's `text_inputs` argument via `previous_result:`.
|
|
1627
|
+
|
|
1628
|
+
### get_dict_value
|
|
1629
|
+
|
|
1630
|
+
Extract a single value from a dictionary result (e.g., a `transformers` pipeline's output) for use in a later step:
|
|
1631
|
+
|
|
1632
|
+
```json
|
|
1633
|
+
{
|
|
1634
|
+
"task": {
|
|
1635
|
+
"command": "get_dict_value",
|
|
1636
|
+
"arguments": {
|
|
1637
|
+
"dict": "previous_result:augment_prompt",
|
|
1638
|
+
"key": "generated_text"
|
|
1639
|
+
}
|
|
1640
|
+
}
|
|
1641
|
+
}
|
|
1642
|
+
```
|
|
1643
|
+
|
|
1644
|
+
| Argument | Required | Description |
|
|
1645
|
+
| -------- | -------- | ----------- |
|
|
1646
|
+
| `dict` | Yes | Dictionary (or `previous_result:` reference) to read from |
|
|
1647
|
+
| `key` | Yes | Key to extract |
|
|
1648
|
+
|
|
1649
|
+
Returns the value at `key`, or `None` if the key is absent.
|
|
1650
|
+
|
|
1651
|
+
### batch_decode_post_process
|
|
1652
|
+
|
|
1653
|
+
Decode generated token IDs and run model-specific post-processing (e.g., Florence-2's task-token parsing), using the processor from an earlier pipeline step:
|
|
1654
|
+
|
|
1655
|
+
```json
|
|
1656
|
+
{
|
|
1657
|
+
"task": {
|
|
1658
|
+
"command": "batch_decode_post_process",
|
|
1659
|
+
"pipeline_reference": "describe_image_processor",
|
|
1660
|
+
"arguments": {
|
|
1661
|
+
"generated_ids": "previous_result:describe_image_model.generated_ids",
|
|
1662
|
+
"task": "<DETAILED_CAPTION>"
|
|
1663
|
+
}
|
|
1664
|
+
}
|
|
1665
|
+
}
|
|
1666
|
+
```
|
|
1667
|
+
|
|
1668
|
+
| Argument | Required | Description |
|
|
1669
|
+
| -------- | -------- | ----------- |
|
|
1670
|
+
| `pipeline_reference` | Yes | Name of an earlier pipeline step whose processor to reuse (sibling of `command`/`arguments`, not inside `arguments`) |
|
|
1671
|
+
| `generated_ids` | Yes | Token IDs to decode (e.g., a model step's `generated_ids` output) |
|
|
1672
|
+
| `task` | Yes | Task token to post-process for (e.g., `<DETAILED_CAPTION>`) |
|
|
1673
|
+
|
|
1674
|
+
Calls `processor.batch_decode(...)` then `processor.post_process_generation(..., task=task)` and returns `parsed_answer[task]`.
|
|
1675
|
+
|
|
1676
|
+
## Multi-Step Example
|
|
1677
|
+
|
|
1678
|
+
Canny edge detection followed by ControlNet generation:
|
|
1679
|
+
|
|
1680
|
+
```json
|
|
1681
|
+
{
|
|
1682
|
+
"steps": [
|
|
1683
|
+
{
|
|
1684
|
+
"name": "edges",
|
|
1685
|
+
"task": {
|
|
1686
|
+
"command": "canny",
|
|
1687
|
+
"arguments": {
|
|
1688
|
+
"image": {
|
|
1689
|
+
"location": "photo.jpg",
|
|
1690
|
+
"low_threshold": 50,
|
|
1691
|
+
"high_threshold": 200
|
|
1692
|
+
}
|
|
1693
|
+
}
|
|
1694
|
+
},
|
|
1695
|
+
"result": { "content_type": "image/jpeg" }
|
|
1696
|
+
},
|
|
1697
|
+
{
|
|
1698
|
+
"name": "generate",
|
|
1699
|
+
"pipeline": {
|
|
1700
|
+
"configuration": {
|
|
1701
|
+
"component_type": "FluxControlPipeline",
|
|
1702
|
+
"offload": "sequential"
|
|
1703
|
+
},
|
|
1704
|
+
"from_pretrained_arguments": {
|
|
1705
|
+
"model_name": "black-forest-labs/FLUX.1-Canny-dev",
|
|
1706
|
+
"torch_dtype": "torch.bfloat16"
|
|
1707
|
+
},
|
|
1708
|
+
"arguments": {
|
|
1709
|
+
"control_image": "previous_result:edges",
|
|
1710
|
+
"prompt": "a watercolor painting",
|
|
1711
|
+
"num_inference_steps": 50
|
|
1712
|
+
}
|
|
1713
|
+
},
|
|
1714
|
+
"result": { "content_type": "image/jpeg" }
|
|
1715
|
+
}
|
|
1716
|
+
]
|
|
1717
|
+
}
|
|
1718
|
+
```
|
|
1719
|
+
|
|
1720
|
+
## Examples
|
|
1721
|
+
|
|
1722
|
+
- [controlnet.json](../workflows/templates/controlnet.json) — Canny edge ControlNet
|
|
1723
|
+
- [controlnet.json](../workflows/templates/controlnet.json) — Depth-guided generation
|
|
1724
|
+
- [qr-code.json](../workflows/templates/qr-code.json) — QR code with artistic ControlNet
|
|
1725
|
+
- [upscale-spandrel.json](../workflows/templates/upscale-spandrel.json) — Spandrel 4x upscale of an existing image
|
|
1726
|
+
- [restore-faces.json](../workflows/templates/restore-faces.json) — Generate portrait + GFPGAN face restoration
|
|
1727
|
+
- [segment.json](../workflows/templates/segment.json) — Text-prompted object segmentation
|
|
1728
|
+
- [segment-and-inpaint.json](../workflows/templates/segment-and-inpaint.json) — Segment + inpaint
|
|
1729
|
+
- [image-to-text.json](../workflows/templates/image-to-text.json) — image captioning with the SmolVLM default
|
|
1730
|
+
- [image-to-text.json](../workflows/templates/image-to-text.json) — VLM captioning with a specific question
|
|
1731
|
+
- [describe-and-regenerate.json](../workflows/templates/describe-and-regenerate.json) — Describe, expand, then regenerate
|
|
1732
|
+
- [interpolate-frames.json](../workflows/templates/interpolate-frames.json) — RIFE frame interpolation
|
|
1733
|
+
- [embed-metadata.json](../workflows/templates/embed-metadata.json) — Embed generation parameters in PNG
|
|
1734
|
+
- [expand-prompt.json](../workflows/templates/expand-prompt.json) — LLM prompt expansion
|
|
1735
|
+
- [expand-prompt.json](../workflows/templates/expand-prompt.json) — Expand prompt + generate image
|
|
1736
|
+
- [upscale-spandrel.json](../workflows/templates/upscale-spandrel.json) — Spandrel upscale of an existing image
|
|
1737
|
+
- [upscale-diffusion.json](../workflows/templates/upscale-diffusion.json) — Diffusion upscale of an existing image
|
|
1738
|
+
- [audio-trim-fade.json](../workflows/templates/audio-trim-fade.json) — Trim a generated track and fade its tail
|
|
1739
|
+
- [generate-speech.json](../workflows/templates/generate-speech.json) — Speak a line with a local text-to-speech model
|
|
1740
|
+
- [voice-timbre-reference.json](../workflows/templates/minimax/voice-timbre-reference.json) — Generate a voice and condition H3's `<Audio 1>` on it
|
|
1741
|
+
- [dissolve-between-shots.json](../workflows/templates/dissolve-between-shots.json) — Dissolve between supplied shots and mix a score under their own audio
|