@sogni-ai/sogni-protocol 1.0.0-alpha.2 → 1.0.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/catalogs/audio-models.json +68 -7
- package/catalogs/quality-presets.json +3 -3
- package/catalogs/seedance-reference-limits.json +35 -0
- package/enums/tool-names.json +2 -0
- package/manifests/composition-tools.json +3 -3
- package/manifests/generation-tools.json +153 -77
- package/manifests/openai-tools.json +140 -65
- package/package.json +1 -1
- package/prompts/tools/animate_photo.json +1 -1
- package/prompts/tools/compose_script.json +1 -1
- package/prompts/tools/compose_workflow.json +1 -1
- package/prompts/tools/compose_workflow_template.json +1 -1
- package/prompts/tools/edit_image.json +1 -1
- package/prompts/tools/enhance_prompt.json +1 -1
- package/prompts/tools/extend_video.json +2 -2
- package/prompts/tools/generate_image.json +2 -2
- package/prompts/tools/generate_video.json +1 -1
- package/prompts/tools/map_assets_for_model.json +1 -1
- package/prompts/tools/replace_video_segment.json +2 -2
- package/prompts/tools/resolve_personas.json +1 -1
- package/prompts/tools/sound_to_video.json +3 -2
- package/prompts/tools/video_to_video.json +2 -2
- package/schemas/agent/intent-input.schema.json +128 -0
- package/schemas/agent/turn-analysis.schema.json +75 -0
- package/schemas/artifacts/artifact-graph.schema.json +42 -0
- package/schemas/artifacts/artifact-node.schema.json +137 -0
- package/schemas/billing/spend-gate.schema.json +151 -0
- package/schemas/billing/workflow-authorization.schema.json +83 -0
- package/schemas/events/run-event.schema.json +122 -0
- package/schemas/tools/animate_photo.schema.json +23 -12
- package/schemas/tools/compose_script.schema.json +1 -1
- package/schemas/tools/compose_workflow.schema.json +2 -2
- package/schemas/tools/compose_workflow_template.schema.json +2 -2
- package/schemas/tools/edit_image.schema.json +8 -7
- package/schemas/tools/enhance_prompt.schema.json +1 -1
- package/schemas/tools/extend_video.schema.json +8 -5
- package/schemas/tools/generate_image.schema.json +12 -9
- package/schemas/tools/generate_music.schema.json +3 -2
- package/schemas/tools/generate_video.schema.json +24 -14
- package/schemas/tools/replace_video_segment.schema.json +5 -2
- package/schemas/tools/sound_to_video.schema.json +16 -8
- package/schemas/tools/tool-metadata.schema.json +78 -0
- package/schemas/tools/upscale_image.schema.json +31 -0
- package/schemas/tools/video_to_video.schema.json +13 -10
- package/schemas/workflows/durable-workflow-run.schema.json +1 -0
- package/version.json +1 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"version": "2026-
|
|
2
|
+
"version": "2026-08-14.2",
|
|
3
3
|
"source": "@sogni-ai/sogni-protocol hosted creative-tools surface (generation + composition)",
|
|
4
4
|
"generatedAt": null,
|
|
5
5
|
"tools": [
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
"type": "function",
|
|
8
8
|
"function": {
|
|
9
9
|
"name": "generate_image",
|
|
10
|
-
"description": "Generate a new image from a text description. Usually this is text-only: do NOT use this tool when the user expects an existing image to be reused or preserved in the result. That includes (a) people from My Personas, and (b) uploaded assets such as logos, brand marks, mascots, product shots, photos, screenshots, sketches, character designs, or other reference images they want carried through. Use edit_image with sourceImageIndex=-1 (or the appropriate generated index) instead. Exception: when the user explicitly requests Z-image, Z Image,
|
|
10
|
+
"description": "Generate a new image from a text description. Usually this is text-only: do NOT use this tool when the user expects an existing image to be reused or preserved in the result. That includes (a) people from My Personas, and (b) uploaded assets such as logos, brand marks, mascots, product shots, photos, screenshots, sketches, character designs, or other reference images they want carried through. Use edit_image with sourceImageIndex=-1 (or the appropriate generated index) instead. Exception: when the user explicitly requests Z-image, Z Image, Z-image Turbo, or Krea 2 Turbo for an uploaded-image enhancement/image-to-image request, use this tool with model=\"z-turbo\", model=\"z-image\", or model=\"krea-2-turbo\", sourceImageIndex=-1, and starting_image_strength because edit_image does not expose those base image-to-image models.",
|
|
11
11
|
"parameters": {
|
|
12
12
|
"type": "object",
|
|
13
13
|
"properties": {
|
|
@@ -21,15 +21,18 @@
|
|
|
21
21
|
"gpt-image-2",
|
|
22
22
|
"z-turbo",
|
|
23
23
|
"z-image",
|
|
24
|
+
"krea-2-turbo",
|
|
25
|
+
"dark-beast-krea2",
|
|
26
|
+
"dark-beast-z-turbo",
|
|
24
27
|
"chroma-v46-flash",
|
|
28
|
+
"chroma1-hd",
|
|
25
29
|
"chroma-detail",
|
|
26
|
-
"flux1-krea",
|
|
27
|
-
"flux2",
|
|
28
30
|
"pony-v7",
|
|
29
31
|
"qwen-2512",
|
|
30
32
|
"qwen-2512-lightning",
|
|
31
33
|
"albedo-xl",
|
|
32
34
|
"animagine-xl",
|
|
35
|
+
"one-obsession-v22",
|
|
33
36
|
"anima-pencil-xl",
|
|
34
37
|
"art-universe-xl",
|
|
35
38
|
"hyphoria-real",
|
|
@@ -41,15 +44,15 @@
|
|
|
41
44
|
"pony-faetality",
|
|
42
45
|
"dreamshaper-xl"
|
|
43
46
|
],
|
|
44
|
-
"description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, or explicitly asks for Z-image/Z-image Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"
|
|
47
|
+
"description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, asks for anime without naming a model, requests permitted NSFW/nudity content, or explicitly asks for Z-image/Z-image Turbo/Krea 2 Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"one-obsession-v22\" when the user asks for an anime or anime-style image and has not named a specific image model. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. Set \"krea-2-turbo\" when the user asks for Krea 2 Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image and Krea 2 Turbo image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"/Qwen image models CANNOT do nudity. For permitted NSFW/nudity content, prefer \"dark-beast-krea2\", then \"dark-beast-z-turbo\"; \"chroma1-hd\", \"pony-v7\", \"chroma-detail\", \"chroma-v46-flash\", and \"z-turbo\" are compatible fallbacks."
|
|
45
48
|
},
|
|
46
49
|
"width": {
|
|
47
50
|
"type": "number",
|
|
48
|
-
"description": "Output image width in pixels. Default: 1024. Supported
|
|
51
|
+
"description": "Output image width in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"1080x1920\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
49
52
|
},
|
|
50
53
|
"height": {
|
|
51
54
|
"type": "number",
|
|
52
|
-
"description": "Output image height in pixels. Default: 1024. Supported
|
|
55
|
+
"description": "Output image height in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"1080x1920\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
53
56
|
},
|
|
54
57
|
"numberOfVariations": {
|
|
55
58
|
"type": "number",
|
|
@@ -63,7 +66,7 @@
|
|
|
63
66
|
},
|
|
64
67
|
"starting_image_strength": {
|
|
65
68
|
"type": "number",
|
|
66
|
-
"description": "Image-to-image strength (0.0-1.0). Only
|
|
69
|
+
"description": "Image-to-image source guidance strength (0.0-1.0). Only set when a source image is available and the model supports img2img. For Z-Image/Z-Image Turbo, Krea 2 Turbo, or source-preserving enhancement requests, use 0.75 with sourceImageIndex so the source image remains a strong guide while allowing higher-resolution reconstruction. Use lower values only when the user explicitly asks for a lighter guide/subtle variation; higher values are more creative and can deviate further from the source."
|
|
67
70
|
},
|
|
68
71
|
"sourceImageIndex": {
|
|
69
72
|
"type": "number",
|
|
@@ -112,7 +115,7 @@
|
|
|
112
115
|
"type": "function",
|
|
113
116
|
"function": {
|
|
114
117
|
"name": "generate_video",
|
|
115
|
-
"description": "Generate a video from text or Seedance multimodal references. LTX 2.3 generates audio natively (dialogue, sounds, ambient music) — describe audio in the prompt. If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23
|
|
118
|
+
"description": "Generate a video from text or Seedance multimodal references. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively (dialogue, sounds, ambient music) — describe audio in the prompt. If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. For non-Seedance syncing to a specific song or audio track, use sound_to_video instead. For non-Seedance animation from a locked source photo, use animate_photo. Do NOT use for My Personas unless generating a Seedance reference-based video — standard persona videos use resolve_personas → edit_image → animate_photo. SEEDANCE DEFAULT: For seedance2, seedance2-mini, seedance2-fast, or seedance2-5, default to exactly one video (4-15s on 2.0/Mini/Fast, 4-30s on seedance2-5) unless the user explicitly asks for multiple separate outputs. Multiple beats, shots, or scene descriptions in one Seedance prompt within the selected model's per-clip limit are still one video. If the user requests one continuous Seedance video longer than 15s, prefer \"seedance2-5\", which renders up to 30s in a single call; beyond 30s (or on 2.0/Mini/Fast) preserve the requested total duration in the prompt/context and let chat orchestration split it into supported segment renders and stitch them instead of clamping it to a short excerpt. Uploaded/generated storyboard, shot-sheet, or trailer-concept images used as Seedance references should become one Seedance generate_video call by default; do not extract panels with edit_image and do not animate the storyboard sheet with LTX unless the user explicitly asks for separate non-Seedance clips. Seedance loose image, video, and audio references go through this tool; do not use animate_photo sourceImageIndex/frameRole/endImageIndex for Seedance. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. If the uploaded audio is the primary sync target, lip-sync target, or requested as sound-to-video/audio-sync, use sound_to_video with videoModel=\"seedance2-mini\" instead of this tool unless the user asks for full Seedance. Use referenceAudioIndices here only when audio is a loose reference under an image/video-anchored Seedance shot. For Seedance, every image — first frame, last frame, or loose reference — is passed through referenceImageIndices (auto-uploaded as referenceImageUrls). Anchor frame intent in the prompt with @Image tags such as \"Use @Image1 as the opening shot reference. Begin the video with a composition, subject placement, lighting, mood, and camera framing that closely match @Image1.\" (or @Image2 as the final shot reference). For seamless-loop or \"first frame and last frame identical\" requests with a single uploaded image, anchor it explicitly as both: \"Use @Image1 as both the first frame and last frame so the video loops cleanly back to the opening composition.\" Assign each useful @Image/@Video/@Audio tag a role. APPROVED STORYBOARD PRODUCTION: When the user asks for a production workflow from an approved storyboard, the chat orchestrator should use the durable CampaignStoryboard contract: render the composite board, audit it, generate per-scene GPT Image 2 keyframes, then render Seedance scene clips and stitch them. Do not replace that with a generic storyboard-reference video unless the user asks for a fast draft. PARTIAL VIDEO EDITS: Do NOT call generate_video to re-render an existing rendered/uploaded video just to change part of it (the bumper, the intro, the end card, a single scene, the last few seconds, etc.). Use replace_video_segment for that — it preserves the unchanged portion, keeps the original audio outside the replaced window, and costs far less. Likewise use extend_video to add new time to the end without rewriting the rest. If the request is vague, ask about vision/mood/style first. Only call once you have clear creative intent.",
|
|
116
119
|
"parameters": {
|
|
117
120
|
"type": "object",
|
|
118
121
|
"properties": {
|
|
@@ -130,60 +133,70 @@
|
|
|
130
133
|
},
|
|
131
134
|
"duration": {
|
|
132
135
|
"type": "number",
|
|
133
|
-
"description": "Video duration in seconds. Default: 5. Range: 2-20. Use when the user explicitly requests a specific length.",
|
|
136
|
+
"description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.5, LTX 2.3, and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30 — only \"seedance2-5\" can use the 16-30s part of this range. MiniMax H3 is quantized to a 17-frame grid at a fixed 24 fps and renders 124-362 frames, so an H3 clip runs 5.17-15.08 seconds and a requested length outside that window snaps to the nearest valid H3 length. Use when the user explicitly requests a specific length.",
|
|
134
137
|
"minimum": 2,
|
|
135
|
-
"maximum":
|
|
138
|
+
"maximum": 30
|
|
136
139
|
},
|
|
137
140
|
"negativePrompt": {
|
|
138
141
|
"type": "string",
|
|
139
|
-
"description": "
|
|
142
|
+
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. All standard LTX 2.5 workflow IDs accept this separate negative prompt. Use this field only when the user explicitly asks to set one. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt. Do not set for MiniMax H3, Seedance, or HappyHorse."
|
|
140
143
|
},
|
|
141
144
|
"videoModel": {
|
|
142
145
|
"type": "string",
|
|
143
146
|
"enum": [
|
|
147
|
+
"ltx25",
|
|
144
148
|
"ltx23",
|
|
145
149
|
"wan22",
|
|
146
150
|
"seedance2",
|
|
147
|
-
"seedance2-
|
|
151
|
+
"seedance2-mini",
|
|
152
|
+
"seedance2-fast",
|
|
153
|
+
"seedance2-5",
|
|
154
|
+
"minimax-h3-t2v",
|
|
155
|
+
"minimax-h3-t2v-turbo",
|
|
156
|
+
"happyhorse-1.1-t2v",
|
|
157
|
+
"happyhorse-1.1-i2v",
|
|
158
|
+
"happyhorse-1.1-r2v",
|
|
159
|
+
"minimax-h3-r2v",
|
|
160
|
+
"minimax-h3-r2v-turbo"
|
|
148
161
|
],
|
|
149
|
-
"description": "Video model. \"
|
|
162
|
+
"description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" only when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft, Mini, or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0, Mini, and Fast accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with no more than 30 reference media files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
|
|
150
163
|
},
|
|
151
164
|
"generateAudio": {
|
|
152
165
|
"type": "boolean",
|
|
153
|
-
"description": "
|
|
166
|
+
"description": "Whether to include generated/native audio for audio-capable models. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. When false, the returned video has no audio track. Not supported by WAN or HappyHorse."
|
|
154
167
|
},
|
|
155
168
|
"referenceImageIndices": {
|
|
156
169
|
"type": "array",
|
|
157
170
|
"items": {
|
|
158
171
|
"type": "number"
|
|
159
172
|
},
|
|
160
|
-
"description": "
|
|
173
|
+
"description": "Image references for Seedance (@Image tags), HappyHorse 1.1 r2v, and MiniMax H3 r2v. Use negative indices for uploaded images (-1 first upload, -2 second upload) and non-negative indices for generated image results. For Seedance, omit by default: uploaded images are auto-forwarded as @Image references. Anchor frame intent in the prompt with @Image tags: \"Use @Image1 as the opening shot reference. Begin the video with a composition, subject placement, lighting, mood, and camera framing that closely match @Image1.\" (or @Image2 as the final shot reference). For seamless-loop or \"first frame and last frame identical\" requests with a single uploaded image, anchor it explicitly as both: \"Use @Image1 as both the first frame and last frame so the video loops cleanly back to the opening composition.\" Do not use animate_photo sourceImageIndex/frameRole/endImageIndex for Seedance. For HappyHorse 1.1 r2v, pass 1-9 image references. For MiniMax H3 r2v, up to 9 images are accepted; at least one image or video reference is required, and H3 references are loose references, not locked frames, and are addressed in the prompt as <Picture 1>, <Picture 2>, and so on in selection order."
|
|
161
174
|
},
|
|
162
175
|
"referenceVideoIndices": {
|
|
163
176
|
"type": "array",
|
|
164
177
|
"items": {
|
|
165
178
|
"type": "number"
|
|
166
179
|
},
|
|
167
|
-
"description": "
|
|
180
|
+
"description": "Optional loose video references for Seedance and MiniMax H3 r2v. Use negative indices for uploaded videos (-1 first uploaded video, -2 second uploaded video) and non-negative indices for generated video results. For Seedance, omit by default: uploaded videos are auto-forwarded as @Video references. Set to choose a subset or include previously generated video URLs. Do not use this for uploaded source-video transforms, upscales, enhancements, restyles, or remasters; use video_to_video with controlMode=\"seedance-v2v\" instead. For MiniMax H3 r2v, up to 3 reference videos (24 fps, 2-15s each, optional soundtrack), addressed as <Video 1>, <Video 2>, and so on in selection order; they can satisfy the required visual reference without an image."
|
|
168
181
|
},
|
|
169
182
|
"referenceAudioIndices": {
|
|
170
183
|
"type": "array",
|
|
171
184
|
"items": {
|
|
172
185
|
"type": "number"
|
|
173
186
|
},
|
|
174
|
-
"description": "
|
|
187
|
+
"description": "Optional loose audio references for Seedance and MiniMax H3 r2v. Use negative indices for uploaded audio files (-1 first uploaded audio, -2 second uploaded audio) and non-negative indices for generated audio results. For Seedance, omit by default: uploaded audio is auto-forwarded as @Audio references when the Seedance request also has an image or video reference. Use this only for loose background, mood, timing, or style references under an image/video-anchored Seedance shot. If the uploaded audio is the primary sync target, lip-sync target, or requested as sound-to-video/audio-sync, use sound_to_video with videoModel=\"seedance2-mini\" instead unless the user asks for full Seedance. Audio-only Seedance requests are unsupported; use sound_to_video for uploaded-audio-only workflows. For MiniMax H3 r2v, up to 3 standalone audio tracks, addressed as <Audio 1>, <Audio 2>, and so on in selection order — a reference video's own soundtrack takes its Audio number before standalone tracks; they supplement a required image or video reference and cannot be the sole input."
|
|
175
188
|
},
|
|
176
189
|
"width": {
|
|
177
190
|
"type": "number",
|
|
178
|
-
"description": "Video width in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Default resolution depends on model and quality tier: LTX Fast about 720p and High/Pro about 1080p; WAN Fast uses 480p short side and High/Pro uses 720p short side. Set width only when the user specifies an exact width or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target, not an instruction to make landscape 1280x720. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override the default media quality. Mappings when orientation is explicit: 480p landscape=854x480, 480p portrait=480x854, 720p landscape=1280x720, 720p portrait=720x1280, 1080p landscape=1920x1080, 1080p portrait=1080x1920, 4K landscape=3840x2160. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
191
|
+
"description": "Video width in pixels. LTX 2.5 and LTX 2.3: 640-3840. WAN: 480-1536. Default resolution depends on model and quality tier: LTX Fast about 720p and High/Pro about 1080p; WAN Fast uses 480p short side and High/Pro uses 720p short side. Set width only when the user specifies an exact width or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target, not an instruction to make landscape 1280x720. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override the default media quality. Mappings when orientation is explicit: 480p landscape=854x480, 480p portrait=480x854, 720p landscape=1280x720, 720p portrait=720x1280, 1080p landscape=1920x1080, 1080p portrait=1080x1920, 4K landscape=3840x2160. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
179
192
|
},
|
|
180
193
|
"height": {
|
|
181
194
|
"type": "number",
|
|
182
|
-
"description": "Video height in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Set height only when the user specifies an exact height or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target; do not convert it to landscape dimensions unless the user says landscape/horizontal/widescreen. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override Default Media Quality, including Pro. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
195
|
+
"description": "Video height in pixels. LTX 2.5 and LTX 2.3: 640-3840. WAN: 480-1536. Set height only when the user specifies an exact height or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target; do not convert it to landscape dimensions unless the user says landscape/horizontal/widescreen. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override Default Media Quality, including Pro. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
183
196
|
},
|
|
184
197
|
"targetResolution": {
|
|
185
198
|
"type": "number",
|
|
186
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
199
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-mini\" vs \"seedance2-fast\" vs \"seedance2-5\"). Seedance 2.0 full supports 4K; Seedance Mini, Fast, and Seedance 2.5 support 480p/720p only, so never set 1080p or 4K for \"seedance2-5\". Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
|
|
187
200
|
},
|
|
188
201
|
"numberOfVariations": {
|
|
189
202
|
"type": "number",
|
|
@@ -197,7 +210,7 @@
|
|
|
197
210
|
},
|
|
198
211
|
"voicePersonaName": {
|
|
199
212
|
"type": "string",
|
|
200
|
-
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one described in the video (e.g. \"David\" narrates a scene featuring Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23."
|
|
213
|
+
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one described in the video (e.g. \"David\" narrates a scene featuring Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
|
|
201
214
|
}
|
|
202
215
|
},
|
|
203
216
|
"required": [
|
|
@@ -242,9 +255,10 @@
|
|
|
242
255
|
"type": "string",
|
|
243
256
|
"enum": [
|
|
244
257
|
"turbo",
|
|
245
|
-
"sft"
|
|
258
|
+
"sft",
|
|
259
|
+
"music3"
|
|
246
260
|
],
|
|
247
|
-
"description": "
|
|
261
|
+
"description": "Music model. \"turbo\" (default): ACE-Step 1.5 Turbo — fast 4-16 step drafts at half cost. \"sft\": ACE-Step 1.5 SFT — experimental, strong lyric handling, 10-200 steps, full cost. \"music3\": MiniMax Music 3 — premium autoregressive composer with the best vocals, lyric adherence and song structure; 30 steps, up to 5 minutes, ~20x turbo cost, and it treats duration as a ceiling (may end the song early at a musical resolution). Use music3 when the user asks for the best quality, realistic vocals, or full songs; otherwise default to \"turbo\"."
|
|
248
262
|
},
|
|
249
263
|
"timesig": {
|
|
250
264
|
"type": "number",
|
|
@@ -273,7 +287,7 @@
|
|
|
273
287
|
"type": "function",
|
|
274
288
|
"function": {
|
|
275
289
|
"name": "edit_image",
|
|
276
|
-
"description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images,
|
|
290
|
+
"description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images, Qwen up to 3 images, and Krea 2 Identity Edit / Dark Beast Krea 2 Identity Edit up to 2 images. Best for style-guided generation, combining elements from multiple images, ANY persona image creation, identity-preserving Krea edits, and any uploaded brand asset reuse — logos, brand marks, mascots, product shots, photos, screenshots, sketches, or character designs the user expects to appear in or guide the result. ALWAYS use this (never generate_image) when persona photos OR uploaded image assets meant for reuse are in context — even if a specific edit model is requested. Exception: explicit Z-image/Z-image Turbo/Krea 2 Turbo uploaded-image enhancement uses generate_image with sourceImageIndex and starting_image_strength because those base image-to-image models are not edit_image models. If a previous edit_image attempt did not preserve the uploaded asset well, stay on edit_image and tighten the prompt or switch model; generate_image has no access to the upload.",
|
|
277
291
|
"parameters": {
|
|
278
292
|
"type": "object",
|
|
279
293
|
"properties": {
|
|
@@ -287,9 +301,10 @@
|
|
|
287
301
|
"gpt-image-2",
|
|
288
302
|
"qwen-lightning",
|
|
289
303
|
"qwen",
|
|
290
|
-
"
|
|
304
|
+
"krea-identity-edit",
|
|
305
|
+
"dark-beast-krea2-identity-edit"
|
|
291
306
|
],
|
|
292
|
-
"description": "
|
|
307
|
+
"description": "The app auto-selects Fast→Qwen Lightning and HQ/Pro→full Qwen only for ordinary identity-neutral edits. REQUIRED IDENTITY DEFAULT: set \"krea-identity-edit\" whenever an edit of a referenced person or character must keep likeness or character identity while changing clothing, hair or makeup, pose or position, face/head/body, background, lighting, or visual style. Infer that semantic intent in any language; never route from keyword or regex matching. Also use it for a non-Pro single-character sheet. This default applies even when the user did not name Krea; an explicitly requested model always wins. Set \"dark-beast-krea2-identity-edit\" only when the user explicitly requests that model, its uncensored/community variant, or dark_beast_krea2_identity_edit_v1_2. Set \"gpt-image-2\" when the user explicitly names GPT/OpenAI/ChatGPT Image, or when precise typography, dense labels, or a professional multi-panel layout is the primary requirement; Pro character sheets may retain GPT Image 2. If GPT Image 2 is unavailable for detail-critical layout work, fall back to full \"qwen\", never \"qwen-lightning\". Krea identity edit models require at least one reference image, accept up to two context images, and work best at 512-2048px. Let the model tier and worker choose current steps, guidance, sampler, scheduler, grounding, and reference-boost defaults; do not send a negative prompt. When Krea is selected, override the generic prompt-length guidance with a concise 1-4 sentence delta instruction; name only the requested change and details that must remain fixed. Put the base scene/image first and an optional person/detail reference second. Z-image, Z-image Turbo, and base Krea 2 Turbo are generate_image img2img models, not edit_image selectors. If the user names another edit/image model, honor it. GPT Image 2 always processes input images at high fidelity; do not set input_fidelity."
|
|
293
308
|
},
|
|
294
309
|
"sourceImageIndex": {
|
|
295
310
|
"type": "number",
|
|
@@ -303,11 +318,11 @@
|
|
|
303
318
|
},
|
|
304
319
|
"width": {
|
|
305
320
|
"type": "number",
|
|
306
|
-
"description": "Output image width in pixels. Defaults to the context image width. Supported
|
|
321
|
+
"description": "Output image width in pixels. Defaults to the context image width. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
307
322
|
},
|
|
308
323
|
"height": {
|
|
309
324
|
"type": "number",
|
|
310
|
-
"description": "Output image height in pixels. Defaults to the context image height. Supported
|
|
325
|
+
"description": "Output image height in pixels. Defaults to the context image height. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
311
326
|
},
|
|
312
327
|
"aspectRatio": {
|
|
313
328
|
"type": "string",
|
|
@@ -430,6 +445,38 @@
|
|
|
430
445
|
}
|
|
431
446
|
}
|
|
432
447
|
},
|
|
448
|
+
{
|
|
449
|
+
"type": "function",
|
|
450
|
+
"function": {
|
|
451
|
+
"name": "upscale_image",
|
|
452
|
+
"description": "Enlarge an existing image with NVIDIA RTX Video Super Resolution while preserving its content, identity, composition, and colors. This is deterministic reconstruction, not a generative edit: it takes no prompt and must not be used for restoration, sharpening requests that imply repainting, object changes, style changes, or creative enhancement. Use it when the user asks to upscale, enlarge, increase resolution, prepare for print, or produce a 2K/4K/6K/8K copy without changing the image.",
|
|
453
|
+
"parameters": {
|
|
454
|
+
"type": "object",
|
|
455
|
+
"properties": {
|
|
456
|
+
"sourceImageIndex": {
|
|
457
|
+
"type": "number",
|
|
458
|
+
"description": "Source image to upscale. Non-negative values select a prior generated result by 0-based index. Negative values select uploads: -1 is the first uploaded image, -2 the second, and so on. If omitted, use the latest generated image, falling back to the first upload."
|
|
459
|
+
},
|
|
460
|
+
"scale": {
|
|
461
|
+
"type": "number",
|
|
462
|
+
"enum": [
|
|
463
|
+
2,
|
|
464
|
+
3,
|
|
465
|
+
4
|
|
466
|
+
],
|
|
467
|
+
"description": "Edge scale multiplier. Use 2 by default. Ignored when targetLongestEdge is supplied. If that scale would leave either aligned output edge below 512px, the tool reports the minimum valid target instead of stretching the image."
|
|
468
|
+
},
|
|
469
|
+
"targetLongestEdge": {
|
|
470
|
+
"type": "number",
|
|
471
|
+
"minimum": 512,
|
|
472
|
+
"maximum": 8192,
|
|
473
|
+
"description": "Optional requested pixel length for the output longest edge, from 512 through 8192. Use 3840 for 4K UHD, 6144 for 6K, 7680 for 8K UHD, or 8192 for an 8K-class maximum. The other edge is calculated automatically so the source aspect ratio is preserved; both aligned output edges must be at least 512px."
|
|
474
|
+
}
|
|
475
|
+
},
|
|
476
|
+
"required": []
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
},
|
|
433
480
|
{
|
|
434
481
|
"type": "function",
|
|
435
482
|
"function": {
|
|
@@ -478,13 +525,13 @@
|
|
|
478
525
|
"type": "function",
|
|
479
526
|
"function": {
|
|
480
527
|
"name": "animate_photo",
|
|
481
|
-
"description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.3 or WAN 2.2. Do NOT use this tool for seedance2 or seedance2-
|
|
528
|
+
"description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.5 by default, LTX 2.3 as rollback, or WAN 2.2. Do NOT use this tool for seedance2, seedance2-mini, seedance2-fast, or seedance2-5. Seedance media references — including Seedance 2.5 first-and-last-frame requests — must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 — always use ltx23 when persona voice is requested (WAN 2.2 does not support voice identity). PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, FIRST generate both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, THEN call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. Never generate the two frames in separate tool calls. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Only call once you have clear creative intent. N-VIDEOS PATTERN — ALWAYS BATCH IN ONE CALL: When the user wants N video versions or a multi-segment stitched non-Seedance video, NEVER call animate_photo N times. Always use sourceImageIndices in a single call so all N projects run in parallel. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips (30s total = 2 clips × 15s), not 5×6s or 6×5s. Two flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image instead of generated source images, use sourceImageIndices=[-1,-1,...] with one -1 per segment. For a long or multi-segment video from a single supplied/uploaded image WITHOUT a requested image/keyframe/version generation stage, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when each clip has DIFFERENT dialogue, jokes, narration, or motion (e.g. \"4 videos where each tells a different joke\"), pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. The fan-out launches all N projects in parallel with their respective per-clip prompts. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame instead.",
|
|
482
529
|
"parameters": {
|
|
483
530
|
"type": "object",
|
|
484
531
|
"properties": {
|
|
485
532
|
"prompt": {
|
|
486
533
|
"type": "string",
|
|
487
|
-
"description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.3 generates audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
|
|
534
|
+
"description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
|
|
488
535
|
},
|
|
489
536
|
"expandPrompt": {
|
|
490
537
|
"type": "boolean",
|
|
@@ -497,22 +544,33 @@
|
|
|
497
544
|
"videoModel": {
|
|
498
545
|
"type": "string",
|
|
499
546
|
"enum": [
|
|
547
|
+
"ltx25",
|
|
500
548
|
"ltx23",
|
|
501
|
-
"wan22"
|
|
549
|
+
"wan22",
|
|
550
|
+
"happyhorse-1.1-i2v",
|
|
551
|
+
"happyhorse-1.1-r2v",
|
|
552
|
+
"minimax-h3-i2v",
|
|
553
|
+
"minimax-h3-i2v-turbo",
|
|
554
|
+
"minimax-h3-flf2v",
|
|
555
|
+
"minimax-h3-flf2v-turbo"
|
|
502
556
|
],
|
|
503
|
-
"description": "Which video model to use. \"
|
|
557
|
+
"description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, seedance2-fast, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
|
|
558
|
+
},
|
|
559
|
+
"generateAudio": {
|
|
560
|
+
"type": "boolean",
|
|
561
|
+
"description": "Whether to include generated/native audio for audio-capable models. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. When false, the returned video has no audio track. Ignored by audio-less WAN."
|
|
504
562
|
},
|
|
505
563
|
"negativePrompt": {
|
|
506
564
|
"type": "string",
|
|
507
|
-
"description": "
|
|
565
|
+
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. All standard LTX 2.5 image and first/last-frame workflows accept this separate negative prompt. Use this field only when the user explicitly asks to set one. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt."
|
|
508
566
|
},
|
|
509
567
|
"duration": {
|
|
510
568
|
"type": "number",
|
|
511
|
-
"description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\").
|
|
569
|
+
"description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Per-model maximum: ltx25 and ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip."
|
|
512
570
|
},
|
|
513
571
|
"targetResolution": {
|
|
514
572
|
"type": "number",
|
|
515
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
573
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
516
574
|
},
|
|
517
575
|
"sourceImageIndex": {
|
|
518
576
|
"type": "number",
|
|
@@ -570,7 +628,7 @@
|
|
|
570
628
|
},
|
|
571
629
|
"voicePersonaName": {
|
|
572
630
|
"type": "string",
|
|
573
|
-
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23."
|
|
631
|
+
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
|
|
574
632
|
}
|
|
575
633
|
},
|
|
576
634
|
"required": [
|
|
@@ -614,13 +672,13 @@
|
|
|
614
672
|
"type": "function",
|
|
615
673
|
"function": {
|
|
616
674
|
"name": "video_to_video",
|
|
617
|
-
"description": "Transform an existing video using
|
|
675
|
+
"description": "Transform an existing video using WAN 2.2 Animate, LTX 2.5 V2V controls by default, LTX 2.3 as rollback, or Seedance V2V when explicitly requested. LTX 2.5 distilled supports canny/pose/depth/detailer/inpaint/outpaint; Dev + Speed LoRA supports canny/pose/depth/detailer. Requires an uploaded video.",
|
|
618
676
|
"parameters": {
|
|
619
677
|
"type": "object",
|
|
620
678
|
"properties": {
|
|
621
679
|
"prompt": {
|
|
622
680
|
"type": "string",
|
|
623
|
-
"description": "Describe the TARGET appearance
|
|
681
|
+
"description": "Describe the TARGET appearance, motion, dialogue, audio, and style in positive present-tense language. For LTX 2.5 (default) or LTX 2.3 rollback canny/depth/pose modes, the source preserves the selected structure or motion, so emphasize style, atmosphere, lighting, texture, color, scale, and pacing. Canny preserves edges; pose preserves skeletal motion; depth preserves 3D layout; detailer should describe the original content with quality qualifiers only. Distilled LTX 2.5 also supports inpaint and outpaint; Dev + Speed LoRA does not. For inpaint, describe only the regenerated region. For outpaint, describe the newly revealed area consistently with the source. For Seedance V2V, use natural prose and describe the target transformation holistically."
|
|
624
682
|
},
|
|
625
683
|
"expandPrompt": {
|
|
626
684
|
"type": "boolean",
|
|
@@ -641,29 +699,32 @@
|
|
|
641
699
|
"detailer",
|
|
642
700
|
"seedance-v2v"
|
|
643
701
|
],
|
|
644
|
-
"description": "How the source video and (optional) reference image interact. Pick by user intent:\n• \"animate-move\" (DEFAULT) — WAN 2.2 Animate Move. Applies camera movement and motion from the source video to the reference image, bringing a still photo to life. Requires sourceImageIndex.\n• \"animate-replace\" — WAN 2.2 Animate Replace. Replaces the subject in the source video with the person/character from the reference image, keeping the video's background and motion. Requires sourceImageIndex.\n• \"canny\" — LTX-2.3 edge-detection control. Best for restyling while preserving exact composition and silhouettes (e.g. \"make this footage look like anime / oil painting / watercolor\"). Use for subjects with crisp edges — people, objects, graphics. Video-only; no reference image needed.\n• \"pose\" — LTX-2.3 skeletal tracking. Best for replacing a person while keeping their motion (e.g. \"turn this dancer into a robot\"). Image optional — if provided, controls appearance; otherwise the prompt drives appearance. Requires person-centric motion.\n• \"depth\" — LTX-2.3 depth-map control. Best for restyling scenes with perspective, camera movement, or volumetric content (landscapes, interiors, camera pans). Preserves 3D spatial layout rather than 2D edges; more forgiving than canny when edges are noisy. Video-only.\n• \"detailer\" — LTX-2.3 quality enhancement. Sharpens detail and texture WITHOUT restyling. The prompt must DESCRIBE THE ORIGINAL scene with quality qualifiers (sharp, clean, high-resolution) — never request content changes, new textures, or a new look. Pick this when the user asks to \"improve quality\", \"enhance\", \"upscale\", or \"sharpen\" without a creative transformation.\n• \"seedance-v2v\" — BytePlus Dreamina Seedance
|
|
702
|
+
"description": "How the source video and (optional) reference image interact. Pick by user intent:\n• \"animate-move\" (DEFAULT) — WAN 2.2 Animate Move. Applies camera movement and motion from the source video to the reference image, bringing a still photo to life. Requires sourceImageIndex.\n• \"animate-replace\" — WAN 2.2 Animate Replace. Replaces the subject in the source video with the person/character from the reference image, keeping the video's background and motion. Requires sourceImageIndex.\n• \"canny\" — LTX-2.3 edge-detection control. Best for restyling while preserving exact composition and silhouettes (e.g. \"make this footage look like anime / oil painting / watercolor\"). Use for subjects with crisp edges — people, objects, graphics. Video-only; no reference image needed.\n• \"pose\" — LTX-2.3 skeletal tracking. Best for replacing a person while keeping their motion (e.g. \"turn this dancer into a robot\"). Image optional — if provided, controls appearance; otherwise the prompt drives appearance. Requires person-centric motion.\n• \"depth\" — LTX-2.3 depth-map control. Best for restyling scenes with perspective, camera movement, or volumetric content (landscapes, interiors, camera pans). Preserves 3D spatial layout rather than 2D edges; more forgiving than canny when edges are noisy. Video-only.\n• \"detailer\" — LTX-2.3 quality enhancement. Sharpens detail and texture WITHOUT restyling. The prompt must DESCRIBE THE ORIGINAL scene with quality qualifiers (sharp, clean, high-resolution) — never request content changes, new textures, or a new look. Pick this when the user asks to \"improve quality\", \"enhance\", \"upscale\", or \"sharpen\" without a creative transformation.\n• \"seedance-v2v\" — BytePlus Dreamina Seedance video-to-video. Use only when the user explicitly asks for Seedance on the uploaded source video, such as Seedance Fast upscale, enhance, remaster, restyle, or transform. High-fidelity quality, native audio, time-coded scene control. Seedance V2V reads @Video1 holistically. Use it for restyling, motion transfer, extension, subject replacement, or scene transformation, and assign @Video1 a clear role such as source clip, camera movement, action timing, edit rhythm, or continuation anchor. Distinct from canny/depth/pose which use control-net constraints — Seedance treats the reference video holistically.\nCanny vs depth: canny preserves silhouettes and fine outlines — pick it for subject-led scenes and graphic restyles. Depth preserves 3D structure — pick it for scenes where the camera moves or spatial layout matters more than edge fidelity. Default: \"animate-move\"."
|
|
645
703
|
},
|
|
646
704
|
"negativePrompt": {
|
|
647
705
|
"type": "string",
|
|
648
|
-
"description": "Non-Seedance only. Optional negative prompt
|
|
706
|
+
"description": "Non-Seedance only. Optional negative prompt supported by every LTX 2.5 and LTX 2.3 video-to-video control/edit template, plus WAN. Do not set when controlMode is seedance-v2v or videoModel is seedance2/seedance2-mini/seedance2-fast/seedance2-5; rewrite user-provided Seedance avoid/ban/no-X requests as positive prompt instructions."
|
|
649
707
|
},
|
|
650
708
|
"videoModel": {
|
|
651
709
|
"type": "string",
|
|
652
710
|
"enum": [
|
|
711
|
+
"ltx25-v2v",
|
|
653
712
|
"ltx23-v2v",
|
|
654
713
|
"wan22-animate",
|
|
655
714
|
"seedance2",
|
|
656
|
-
"seedance2-
|
|
715
|
+
"seedance2-mini",
|
|
716
|
+
"seedance2-fast",
|
|
717
|
+
"seedance2-5"
|
|
657
718
|
],
|
|
658
|
-
"description": "Model selector for this video-to-video request. Usually omit; controlMode chooses the non-Seedance model. For controlMode=\"seedance-v2v\", use \"seedance2-
|
|
719
|
+
"description": "Model selector for this video-to-video request. Usually omit; controlMode chooses the non-Seedance model. \"ltx25-v2v\" is the default LTX 2.5 control path; \"ltx23-v2v\" remains the rollback selector. For controlMode=\"seedance-v2v\", Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance V2V unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro or from 480p/720p resolution requests alone. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
|
|
659
720
|
},
|
|
660
721
|
"generateAudio": {
|
|
661
722
|
"type": "boolean",
|
|
662
|
-
"description": "
|
|
723
|
+
"description": "Whether the final video should include generated or retained audio. Omit to include audio by default; set false when the user asks for silent output or no audio. When false, the returned video has no audio track."
|
|
663
724
|
},
|
|
664
725
|
"targetResolution": {
|
|
665
726
|
"type": "number",
|
|
666
|
-
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
727
|
+
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V Mini, Fast, and Seedance 2.5 support 480p and 720p only, so never set 1080p or 4K for \"seedance2-5\". Preserve the source video shape instead of forcing landscape pixels."
|
|
667
728
|
},
|
|
668
729
|
"sourceImageIndex": {
|
|
669
730
|
"type": "number",
|
|
@@ -671,9 +732,9 @@
|
|
|
671
732
|
},
|
|
672
733
|
"duration": {
|
|
673
734
|
"type": "number",
|
|
674
|
-
"description": "Output video duration in seconds. Range: 2-20 for WAN/LTX modes and 4-
|
|
735
|
+
"description": "Output video duration in seconds. Range: 2-20 for WAN/LTX modes, 4-15 for controlMode=\"seedance-v2v\" on \"seedance2\"/\"seedance2-mini\"/\"seedance2-fast\", and 4-30 for controlMode=\"seedance-v2v\" on \"seedance2-5\". If omitted, the tool matches the uploaded source video duration when available (capped to the selected model range); otherwise it falls back to 10s for WAN Animate Move/Replace and 5s for LTX-2.3/Seedance modes. For long stitched/bulk WAN Animate Move/Replace work with no explicit per-clip length, prefer about 10s clips rather than 5s chunks. Only pass this when the user explicitly requests a different length.",
|
|
675
736
|
"minimum": 2,
|
|
676
|
-
"maximum":
|
|
737
|
+
"maximum": 30
|
|
677
738
|
},
|
|
678
739
|
"numberOfVariations": {
|
|
679
740
|
"type": "number",
|
|
@@ -864,25 +925,29 @@
|
|
|
864
925
|
"type": "function",
|
|
865
926
|
"function": {
|
|
866
927
|
"name": "sound_to_video",
|
|
867
|
-
"description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx23-ia2v; for lip-sync with a face image, use wan-s2v; if no image, use ltx23-a2v. If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.3 generates audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video
|
|
928
|
+
"description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx25-ia2v by default (ltx23-ia2v is rollback); for lip-sync with a face image, use wan-s2v; if no image, use ltx25-a2v by default (ltx23-a2v is rollback). If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video with videoModel=\"ltx23\" because LTX 2.5 has no compatible ID-LoRA. LONG AUDIO ON SEEDANCE: Seedance 2.0, Mini, and Fast cap each clip at 15s; Seedance 2.5 caps each clip at 30s, so prefer \"seedance2-5\" for 16-30s audio instead of splitting. When uploaded audio exceeds the selected Seedance model's per-clip cap, do NOT clamp and drop the rest — split the run into multiple sound_to_video calls in the same turn using 15s segments for seedance2/seedance2-mini/seedance2-fast or 30s segments for seedance2-5, then finish with a single stitch_video call referencing the resulting clip indices in order with audioIndex pointing at the same uploaded audio so the stitched output carries the full original soundtrack. LTX/WAN models accept up to 20s per clip, so single-call is fine for them.",
|
|
868
929
|
"parameters": {
|
|
869
930
|
"type": "object",
|
|
870
931
|
"properties": {
|
|
871
932
|
"prompt": {
|
|
872
933
|
"type": "string",
|
|
873
|
-
"description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.3 image+audio mode, do not re-describe static details already visible in the reference image; focus on motion, action, camera, and how the image responds to the audio.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main visual beat + 1 simple camera move. Around 10s: 2-3 clear beats + 1 camera move. >10s: up to 4 beats in clear sequence. Let the audio define timing, but avoid stacking subject, camera, and environment motion in short clips.\n\nBLOCKING: Direct layout when it affects the shot: left/right placement, foreground/background, facing direction, and relative distance between subjects.\n\nLIP-SYNC: Shot framing, speaker's appearance and setting, physical performance synced to audio — gestures, expressions, jaw movement between phrases. Include acting beats.\n\nMUSIC VISUALIZATION: Visual style, environment, and how elements react to rhythm and energy.\n\nAUDIO-REACTIVE: Motion and visual changes that correspond to sounds in the track.\n\nLTX VOCABULARY: camera (tracking, dolly, pan, tilt, handheld, static frame), lighting/atmosphere (golden hour, neon glow, dramatic shadows, fog, rain, smoke, reflections), scale/pacing (expansive, epic, intimate, claustrophobic, slow motion, time-lapse, lingering shot, continuous shot), style/genre (film noir, painterly, cyberpunk, stop-motion, claymation, 2D/3D animation, hand-drawn, fantasy, thriller, experimental film).\n\nAVOID: Vague prompts, too many competing visual elements, abstract descriptions without visible behavior, rigid numeric constraints, readable text or logos. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the visual interpretation while keeping audio sync intent consistent. Example: \"{abstract neon visualization|nature scene with swaying trees|urban street with rain} synced to the beat\"."
|
|
934
|
+
"description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.5 or LTX 2.3 image+audio mode, do not re-describe static details already visible in the reference image; focus on motion, action, camera, and how the image responds to the audio.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main visual beat + 1 simple camera move. Around 10s: 2-3 clear beats + 1 camera move. >10s: up to 4 beats in clear sequence. Let the audio define timing, but avoid stacking subject, camera, and environment motion in short clips.\n\nBLOCKING: Direct layout when it affects the shot: left/right placement, foreground/background, facing direction, and relative distance between subjects.\n\nLIP-SYNC: Shot framing, speaker's appearance and setting, physical performance synced to audio — gestures, expressions, jaw movement between phrases. Include acting beats.\n\nMUSIC VISUALIZATION: Visual style, environment, and how elements react to rhythm and energy.\n\nAUDIO-REACTIVE: Motion and visual changes that correspond to sounds in the track.\n\nLTX VOCABULARY: camera (tracking, dolly, pan, tilt, handheld, static frame), lighting/atmosphere (golden hour, neon glow, dramatic shadows, fog, rain, smoke, reflections), scale/pacing (expansive, epic, intimate, claustrophobic, slow motion, time-lapse, lingering shot, continuous shot), style/genre (film noir, painterly, cyberpunk, stop-motion, claymation, 2D/3D animation, hand-drawn, fantasy, thriller, experimental film).\n\nAVOID: Vague prompts, too many competing visual elements, abstract descriptions without visible behavior, rigid numeric constraints, readable text or logos. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the visual interpretation while keeping audio sync intent consistent. Example: \"{abstract neon visualization|nature scene with swaying trees|urban street with rain} synced to the beat\"."
|
|
874
935
|
},
|
|
875
936
|
"expandPrompt": {
|
|
876
937
|
"type": "boolean",
|
|
877
938
|
"description": "Seedance only. Whether to run the shared Seedance prompt shaper before dispatch. Defaults to true; set false only when the user explicitly asks to submit the compact prompt directly or not modify the prompt."
|
|
878
939
|
},
|
|
940
|
+
"negativePrompt": {
|
|
941
|
+
"type": "string",
|
|
942
|
+
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. The LTX A2V and IA2V workflows accept this separate negative prompt. Use it only when the user explicitly asks to set one. Do not set for Seedance."
|
|
943
|
+
},
|
|
879
944
|
"audioSourceIndex": {
|
|
880
945
|
"type": "number",
|
|
881
946
|
"description": "Index of the uploaded audio file to use (0-based, from uploaded files list). If only one audio file is uploaded, use 0. If no audio was uploaded but generate_music was used earlier, omit this — the tool will automatically find the generated audio."
|
|
882
947
|
},
|
|
883
948
|
"sourceImageIndex": {
|
|
884
949
|
"type": "number",
|
|
885
|
-
"description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.3 A2V), this is optional — omit it to generate video purely from text + audio."
|
|
950
|
+
"description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.5 or LTX 2.3 A2V), this is optional — omit it to generate video purely from text + audio."
|
|
886
951
|
},
|
|
887
952
|
"audioStart": {
|
|
888
953
|
"type": "number",
|
|
@@ -891,24 +956,28 @@
|
|
|
891
956
|
},
|
|
892
957
|
"duration": {
|
|
893
958
|
"type": "number",
|
|
894
|
-
"description": "Video duration in seconds. Default: 5. Range: 2-20. For music videos, use the MAXIMUM duration (20) since the audio is always longer than the video limit. Use when the user explicitly requests a specific length.",
|
|
959
|
+
"description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.5, LTX 2.3, and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30. For music videos, use the MAXIMUM duration the selected model allows (20 for LTX/WAN, 30 for \"seedance2-5\") since the audio is always longer than the video limit. Use when the user explicitly requests a specific length.",
|
|
895
960
|
"minimum": 2,
|
|
896
|
-
"maximum":
|
|
961
|
+
"maximum": 30
|
|
897
962
|
},
|
|
898
963
|
"videoModel": {
|
|
899
964
|
"type": "string",
|
|
900
965
|
"enum": [
|
|
901
966
|
"wan-s2v",
|
|
902
967
|
"seedance2",
|
|
968
|
+
"seedance2-mini",
|
|
903
969
|
"seedance2-fast",
|
|
970
|
+
"seedance2-5",
|
|
971
|
+
"ltx25-ia2v",
|
|
972
|
+
"ltx25-a2v",
|
|
904
973
|
"ltx23-ia2v",
|
|
905
974
|
"ltx23-a2v"
|
|
906
975
|
],
|
|
907
|
-
"description": "Video model. \"
|
|
976
|
+
"description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast/HQ use official distilled INT8 and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. \"seedance2-fast\": legacy Seedance 2.0 Fast. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance unless the user explicitly says Seedance Fast, pick \"seedance2-fast\" when the user says Seedance Fast / seedance-fast, and pick \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
|
|
908
977
|
},
|
|
909
978
|
"generateAudio": {
|
|
910
979
|
"type": "boolean",
|
|
911
|
-
"description": "
|
|
980
|
+
"description": "Whether the final video should include audio. Omit to include audio by default; set false when the user asks for silent output or no audio. When false, the returned video has no audio track; the reference audio is still required and still drives generation."
|
|
912
981
|
},
|
|
913
982
|
"numberOfVariations": {
|
|
914
983
|
"type": "number",
|
|
@@ -918,7 +987,7 @@
|
|
|
918
987
|
},
|
|
919
988
|
"targetResolution": {
|
|
920
989
|
"type": "number",
|
|
921
|
-
"description": "Short-side video resolution target in pixels. Use
|
|
990
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
922
991
|
},
|
|
923
992
|
"aspectRatio": {
|
|
924
993
|
"type": "string",
|
|
@@ -935,7 +1004,7 @@
|
|
|
935
1004
|
"type": "function",
|
|
936
1005
|
"function": {
|
|
937
1006
|
"name": "extend_video",
|
|
938
|
-
"description": "Extend a video by adding new time to the end. Works on BOTH videos previously rendered in this session AND user-uploaded videos — set videoIndex to a negative number (e.g. -1) to target an uploaded video when no prior render exists. The base video is auto-selected from the most recent video in this session unless videoIndex is set. For LTX-2.3 base clips, the tool extracts the last frame and renders an image-to-video continuation. For Seedance base clips, the tool extracts a trailing reference segment and renders a video-to-video continuation. Returns both the standalone new segment and a spliced composite (base + new segment). Use when the user asks to \"make it longer\", \"extend the video\", \"add another N seconds\", \"continue the scene\", \"add an outro/bumper to the end\", etc. Prefer this over generate_image+animate_photo+stitch_video for \"add a bumper/outro to this video\" — extend_video preserves the original base bytes, audio, and timing instead of re-encoding them. Do not use this tool to render fresh videos from scratch — call generate_video or animate_photo for that. Output durations follow each model's native limits (LTX 2-20s, Seedance 4-15s) for the new segment alone.",
|
|
1007
|
+
"description": "Extend a video by adding new time to the end. Works on BOTH videos previously rendered in this session AND user-uploaded videos — set videoIndex to a negative number (e.g. -1) to target an uploaded video when no prior render exists. The base video is auto-selected from the most recent video in this session unless videoIndex is set. For LTX-2.3 base clips, the tool extracts the last frame and renders an image-to-video continuation. For Seedance base clips, the tool extracts a trailing reference segment and renders a video-to-video continuation. Returns both the standalone new segment and a spliced composite (base + new segment). Use when the user asks to \"make it longer\", \"extend the video\", \"add another N seconds\", \"continue the scene\", \"add an outro/bumper to the end\", etc. Prefer this over generate_image+animate_photo+stitch_video for \"add a bumper/outro to this video\" — extend_video preserves the original base bytes, audio, and timing instead of re-encoding them. Do not use this tool to render fresh videos from scratch — call generate_video or animate_photo for that. Output durations follow each model's native limits (LTX 2-20s, Seedance 2.0/Mini/Fast 4-15s, Seedance 2.5 4-30s) for the new segment alone.",
|
|
939
1008
|
"parameters": {
|
|
940
1009
|
"type": "object",
|
|
941
1010
|
"properties": {
|
|
@@ -945,9 +1014,9 @@
|
|
|
945
1014
|
},
|
|
946
1015
|
"duration": {
|
|
947
1016
|
"type": "number",
|
|
948
|
-
"description": "Length in seconds of the new appended segment (NOT total final length). LTX 2-20, Seedance 4-15. Default: 5.",
|
|
1017
|
+
"description": "Length in seconds of the new appended segment (NOT total final length). LTX 2-20, Seedance 2.0/Mini/Fast 4-15, Seedance 2.5 4-30. Default: 5.",
|
|
949
1018
|
"minimum": 2,
|
|
950
|
-
"maximum":
|
|
1019
|
+
"maximum": 30
|
|
951
1020
|
},
|
|
952
1021
|
"videoIndex": {
|
|
953
1022
|
"type": "number",
|
|
@@ -957,11 +1026,14 @@
|
|
|
957
1026
|
"type": "string",
|
|
958
1027
|
"enum": [
|
|
959
1028
|
"auto",
|
|
1029
|
+
"ltx25",
|
|
960
1030
|
"ltx23",
|
|
961
1031
|
"seedance2",
|
|
962
|
-
"seedance2-
|
|
1032
|
+
"seedance2-mini",
|
|
1033
|
+
"seedance2-fast",
|
|
1034
|
+
"seedance2-5"
|
|
963
1035
|
],
|
|
964
|
-
"description": "Which model to use for the new segment. Default: \"auto\" —
|
|
1036
|
+
"description": "Which model to use for the new segment. Default: \"auto\" — preserve Seedance for a Seedance base and otherwise use LTX 2.5. Use ltx23 only for explicit rollback. \"seedance2-5\" supports 480p/720p and 4-30s of new footage at 24 fps."
|
|
965
1037
|
},
|
|
966
1038
|
"keepOriginalAudio": {
|
|
967
1039
|
"type": "boolean",
|
|
@@ -1016,12 +1088,15 @@
|
|
|
1016
1088
|
"type": "string",
|
|
1017
1089
|
"enum": [
|
|
1018
1090
|
"auto",
|
|
1091
|
+
"ltx25",
|
|
1019
1092
|
"ltx23",
|
|
1020
1093
|
"wan22",
|
|
1021
1094
|
"seedance2",
|
|
1022
|
-
"seedance2-
|
|
1095
|
+
"seedance2-mini",
|
|
1096
|
+
"seedance2-fast",
|
|
1097
|
+
"seedance2-5"
|
|
1023
1098
|
],
|
|
1024
|
-
"description": "Which model to use for the new segment. Default: \"auto\" —
|
|
1099
|
+
"description": "Which model to use for the new segment. Default: \"auto\" — preserve Seedance or WAN for matching base clips and otherwise use LTX 2.5. Use ltx23 only for explicit rollback. \"seedance2-5\" supports 480p/720p and 4-30s replacement windows at 24 fps."
|
|
1025
1100
|
},
|
|
1026
1101
|
"keepOriginalAudio": {
|
|
1027
1102
|
"type": "boolean",
|
|
@@ -1270,7 +1345,7 @@
|
|
|
1270
1345
|
},
|
|
1271
1346
|
"destination_model": {
|
|
1272
1347
|
"type": "string",
|
|
1273
|
-
"description": "Optional destination model selector, such as seedance2, ltx23, wan22,
|
|
1348
|
+
"description": "Optional destination model selector, such as seedance2, ltx23, wan22, gpt-image-2, or sdxl."
|
|
1274
1349
|
},
|
|
1275
1350
|
"destination_tool": {
|
|
1276
1351
|
"type": "string",
|
|
@@ -1574,7 +1649,7 @@
|
|
|
1574
1649
|
"properties": {
|
|
1575
1650
|
"image": {
|
|
1576
1651
|
"type": "string",
|
|
1577
|
-
"description": "Preferred image model (e.g., '
|
|
1652
|
+
"description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')."
|
|
1578
1653
|
},
|
|
1579
1654
|
"video": {
|
|
1580
1655
|
"type": "string",
|
|
@@ -1777,7 +1852,7 @@
|
|
|
1777
1852
|
"properties": {
|
|
1778
1853
|
"image": {
|
|
1779
1854
|
"type": "string",
|
|
1780
|
-
"description": "Preferred image model (e.g., '
|
|
1855
|
+
"description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')."
|
|
1781
1856
|
},
|
|
1782
1857
|
"video": {
|
|
1783
1858
|
"type": "string",
|