@sogni-ai/sogni-protocol 1.0.0-alpha.2 → 1.0.0-alpha.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/catalogs/audio-models.json +68 -7
- package/catalogs/quality-presets.json +3 -3
- package/catalogs/seedance-reference-limits.json +35 -0
- package/enums/tool-names.json +2 -0
- package/manifests/composition-tools.json +3 -3
- package/manifests/generation-tools.json +153 -77
- package/manifests/openai-tools.json +140 -65
- package/package.json +1 -1
- package/prompts/tools/animate_photo.json +1 -1
- package/prompts/tools/compose_script.json +1 -1
- package/prompts/tools/compose_workflow.json +1 -1
- package/prompts/tools/compose_workflow_template.json +1 -1
- package/prompts/tools/edit_image.json +1 -1
- package/prompts/tools/enhance_prompt.json +1 -1
- package/prompts/tools/extend_video.json +2 -2
- package/prompts/tools/generate_image.json +2 -2
- package/prompts/tools/generate_video.json +1 -1
- package/prompts/tools/map_assets_for_model.json +1 -1
- package/prompts/tools/replace_video_segment.json +2 -2
- package/prompts/tools/resolve_personas.json +1 -1
- package/prompts/tools/sound_to_video.json +3 -2
- package/prompts/tools/video_to_video.json +2 -2
- package/schemas/agent/intent-input.schema.json +128 -0
- package/schemas/agent/turn-analysis.schema.json +75 -0
- package/schemas/artifacts/artifact-graph.schema.json +42 -0
- package/schemas/artifacts/artifact-node.schema.json +137 -0
- package/schemas/billing/spend-gate.schema.json +151 -0
- package/schemas/billing/workflow-authorization.schema.json +83 -0
- package/schemas/events/run-event.schema.json +122 -0
- package/schemas/tools/animate_photo.schema.json +23 -12
- package/schemas/tools/compose_script.schema.json +1 -1
- package/schemas/tools/compose_workflow.schema.json +2 -2
- package/schemas/tools/compose_workflow_template.schema.json +2 -2
- package/schemas/tools/edit_image.schema.json +8 -7
- package/schemas/tools/enhance_prompt.schema.json +1 -1
- package/schemas/tools/extend_video.schema.json +8 -5
- package/schemas/tools/generate_image.schema.json +12 -9
- package/schemas/tools/generate_music.schema.json +3 -2
- package/schemas/tools/generate_video.schema.json +24 -14
- package/schemas/tools/replace_video_segment.schema.json +5 -2
- package/schemas/tools/sound_to_video.schema.json +16 -8
- package/schemas/tools/tool-metadata.schema.json +78 -0
- package/schemas/tools/upscale_image.schema.json +31 -0
- package/schemas/tools/video_to_video.schema.json +13 -10
- package/schemas/workflows/durable-workflow-run.schema.json +1 -0
- package/version.json +1 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"version": "2026-
|
|
2
|
+
"version": "2026-08-14.2",
|
|
3
3
|
"source": "sogni-creative-agent/src/tools/definitions/*/definition.ts",
|
|
4
4
|
"schemaRefs": {
|
|
5
5
|
"generate_image": "../schemas/tools/generate_image.schema.json",
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
"edit_image": "../schemas/tools/edit_image.schema.json",
|
|
9
9
|
"apply_style": "../schemas/tools/apply_style.schema.json",
|
|
10
10
|
"restore_photo": "../schemas/tools/restore_photo.schema.json",
|
|
11
|
+
"upscale_image": "../schemas/tools/upscale_image.schema.json",
|
|
11
12
|
"refine_result": "../schemas/tools/refine_result.schema.json",
|
|
12
13
|
"animate_photo": "../schemas/tools/animate_photo.schema.json",
|
|
13
14
|
"change_angle": "../schemas/tools/change_angle.schema.json",
|
|
@@ -26,13 +27,13 @@
|
|
|
26
27
|
"type": "function",
|
|
27
28
|
"function": {
|
|
28
29
|
"name": "generate_image",
|
|
29
|
-
"description": "Generate a new image from a text description. Usually this is text-only: do NOT use this tool when the user expects an existing image to be reused or preserved in the result. That includes (a) people from My Personas, and (b) uploaded assets such as logos, brand marks, mascots, product shots, photos, screenshots, sketches, character designs, or other reference images they want carried through. Use edit_image with sourceImageIndex=-1 (or the appropriate generated index) instead. Exception: when the user explicitly requests Z-image, Z Image,
|
|
30
|
+
"description": "Generate a new image from a text description. Usually this is text-only: do NOT use this tool when the user expects an existing image to be reused or preserved in the result. That includes (a) people from My Personas, and (b) uploaded assets such as logos, brand marks, mascots, product shots, photos, screenshots, sketches, character designs, or other reference images they want carried through. Use edit_image with sourceImageIndex=-1 (or the appropriate generated index) instead. Exception: when the user explicitly requests Z-image, Z Image, Z-image Turbo, or Krea 2 Turbo for an uploaded-image enhancement/image-to-image request, use this tool with model=\"z-turbo\", model=\"z-image\", or model=\"krea-2-turbo\", sourceImageIndex=-1, and starting_image_strength because edit_image does not expose those base image-to-image models.",
|
|
30
31
|
"parameters": {
|
|
31
32
|
"type": "object",
|
|
32
33
|
"properties": {
|
|
33
34
|
"prompt": {
|
|
34
35
|
"type": "string",
|
|
35
|
-
"description": "Text description of the image (50-200 words). POSITIVE phrasing only. Be specific and vivid — reference real artists, franchises, and aesthetics by name.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing.\n\nPROMPT ORDER (follow this structure): [SUBJECT] → [ATTRIBUTES] → [ACTION/POSE] → [CAMERA/FRAMING] → [ENVIRONMENT] → [LIGHTING] → [STYLE/MEDIUM] → [MATERIALS/TEXTURES] → [SECONDARY DETAILS].
|
|
36
|
+
"description": "Text description of the image (50-200 words). POSITIVE phrasing only. Be specific and vivid — reference real artists, franchises, and aesthetics by name.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing.\n\nPROMPT ORDER (follow this structure by default): [SUBJECT] → [ATTRIBUTES] → [ACTION/POSE] → [CAMERA/FRAMING] → [ENVIRONMENT] → [LIGHTING] → [STYLE/MEDIUM] → [MATERIALS/TEXTURES] → [SECONDARY DETAILS]. Lead with the main subject and its concrete, observable attributes unless the user explicitly asks for mood, atmosphere, or another prompt shape first. Put the most visually decisive details early.\n\nSPECIFICITY: Use concrete nouns and observable adjectives (\"weathered leather jacket\", not \"cool outfit\"). Specify framing (close-up, medium shot, full body, wide shot), angle (eye level, low angle, high angle, overhead), lighting type (\"soft overcast daylight\", \"warm golden-hour sunlight\", \"moody neon spill with deep shadows\"), and medium/style (\"photorealistic editorial photography\", \"cinematic still frame\", \"clean anime illustration\"). Include materials and textures when relevant (\"brushed aluminum\", \"wet asphalt reflections\", \"heavy wool texture\").\n\nDEFAULTS (fill in when user is underspecified): Framing: medium shot for portraits, wide shot for environments, full-body for fashion/outfits. Angle: eye level unless dramatic perspective requested. Lighting: soft natural light for realism, clean studio light for product shots. Style: photorealistic for realistic models, matching the model's native style for stylized models (e.g. anime illustration for pony/animagine). Reference real artists and franchises by name (\"in the style of Monet's Water Lilies\", \"Wes Anderson symmetrical pastel composition\", \"cyberpunk Blade Runner neon city\", \"shot on 85mm f/1.4 with shallow depth of field\").\n\nAVOID: Starting with abstract mood words alone. Burying the subject after a long style preamble. Stacking incompatible styles. Overloading with competing focal points. Vague phrases like \"very cool\" or \"epic vibes\".\n\nCHARACTER / MASCOT SHEETS: When the user asks for a character sheet, mascot sheet, model sheet, turnaround, expression sheet, or reusable character reference board, create ONE comprehensive professional reference-board image, not separate variations. Include a large hero pose, front / 3/4 / side / back turnaround views, an expression row, action/personality poses, accessories or props, color palette swatches, and compact notes such as personality, fun facts, or brand usage when appropriate. Preserve exact user-provided brand names, slogans, logo text, and requested copy verbatim; incidental tiny notes may be generated by the image model if the user did not provide exact wording. Keep the character consistent across every panel and use clean readable typography.\n\nBATCH VARIATIONS: When numberOfVariations > 1, the prompt describes one output image. Do not mention counts, \"versions\", \"different\", or \"multiple\" in the prompt text unless the user explicitly wants those words visible in the image. Do not describe multiple copies or duplicates of the subject in a single image unless the user asked for a collage, grid, or side-by-side composition. Use Dynamic Prompt syntax to vary one dimension across separate images. Example: user asks \"4 cats in different spots\" → numberOfVariations=4, prompt=\"a black cat {lounging in a sunlit window|prowling through autumn leaves|sitting on a vintage bookshelf|curled up by a fireplace}\" — each output is one cat in one spot. Vary setting, style, lighting, expression, or composition; preserve what the user specified. Preserve any requested orientation, aspect ratio, or exact pixel dimensions across every variation.\n\nSELECTION-GATED IMAGE STAGES: If the user asks for multiple image options/takes/versions and says they will pick one before a later dance, animation, or video, this tool call is still the first step. Generate the complete image batch now with the exact requested count, Dynamic Prompt options for each output, and the final video/image aspect ratio. Do not ask the user to choose before the images exist, and do not call video tools until after the user selects an image.\n\nLINKED VARIANTS: If multiple details must stay paired per output — visual style, outfit, label text, symbol, setting, character, prop, location, or before/after keyframe details — use ONE top-level Dynamic Prompt branch with one complete prompt per output. Do NOT use separate Dynamic Prompt groups for details that must stay together; unpaired groups can mix attributes. If the user asks for per-variant facial, identity, or appearance changes, repeat that guidance inside EVERY option. When the user names a subject or character, write that name or stable role inside every Dynamic Prompt option; a shared prefix outside the branch is not enough because each option must stand alone. Correct shape: \"{full prompt for variant 1 with all paired details|full prompt for variant 2 with all paired details|...}\".\n\nSCREENPLAY / STORYBOARD BATCHES: For multi-scene commercials, storyboards, or shot lists, numberOfVariations should equal the scene count and the prompt should be a single top-level dynamic branch containing one full scene prompt per option, e.g. \"{scene 1 full prompt|scene 2 full prompt|scene 3 full prompt}\". This is the required way to batch scenes with materially different content while still rendering one image per scene. If recurring characters appear, use stable character names and repeat the same visual anchors in every scene option where they appear (age range, build, hairstyle, outfit silhouette, color palette, signature prop/accessory, posture). Do not rename, merge, redesign, or drift characters between scene keyframes unless the user asks. Include speaker-tagged dialogue details when dialogue affects the keyframe, e.g. CHARACTER: \"We made it.\" Do not set numberOfVariations=N with only scene 1's prompt; that creates N duplicate versions of scene 1, not N scenes. If the scene count is 16 or fewer, keep it in one call unless the user explicitly asks for separate projects or per-output settings require separate calls.\n\nCOMPOSITE GPT IMAGE 2 STORYBOARD SHEETS: When numberOfVariations=1 and the user asks for one composite video storyboard/keyframe sheet, the prompt must be a compiled storyboard prompt, not a concept summary. Include a SCENES: section with exactly the requested number of concrete entries named SCENE_01, SCENE_02, etc. Every scene entry must include Visual/Action, Camera/Motion, Dialogue/VO (or [no dialogue]), Audio/SFX, and any visible text or reference usage for that scene. Do not provide only the source brief or generic layout instructions; malformed compiled storyboard prompts are blocked by quality audit.\n\nVIDEO KEYFRAMES: When generating images intended as first+last frames for video (animate_photo with frameRole=\"both\"), use numberOfVariations=2 with Dynamic Prompts to create both frames in one call. Make each frame a distinct scene that creates a compelling transition. The video handler will inspect both generated frames and build a scene-aware transition prompt, so focus this image prompt on producing strong start/end visuals. Example: \"a serene lake {at dawn with mist rising and soft pink sky|at dusk with fireflies and deep blue twilight}\"."
|
|
36
37
|
},
|
|
37
38
|
"model": {
|
|
38
39
|
"type": "string",
|
|
@@ -40,15 +41,18 @@
|
|
|
40
41
|
"gpt-image-2",
|
|
41
42
|
"z-turbo",
|
|
42
43
|
"z-image",
|
|
44
|
+
"krea-2-turbo",
|
|
45
|
+
"dark-beast-krea2",
|
|
46
|
+
"dark-beast-z-turbo",
|
|
43
47
|
"chroma-v46-flash",
|
|
48
|
+
"chroma1-hd",
|
|
44
49
|
"chroma-detail",
|
|
45
|
-
"flux1-krea",
|
|
46
|
-
"flux2",
|
|
47
50
|
"pony-v7",
|
|
48
51
|
"qwen-2512",
|
|
49
52
|
"qwen-2512-lightning",
|
|
50
53
|
"albedo-xl",
|
|
51
54
|
"animagine-xl",
|
|
55
|
+
"one-obsession-v22",
|
|
52
56
|
"anima-pencil-xl",
|
|
53
57
|
"art-universe-xl",
|
|
54
58
|
"hyphoria-real",
|
|
@@ -60,19 +64,19 @@
|
|
|
60
64
|
"pony-faetality",
|
|
61
65
|
"dreamshaper-xl"
|
|
62
66
|
],
|
|
63
|
-
"description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, or explicitly asks for Z-image/Z-image Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"
|
|
67
|
+
"description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, asks for anime without naming a model, requests permitted NSFW/nudity content, or explicitly asks for Z-image/Z-image Turbo/Krea 2 Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"one-obsession-v22\" when the user asks for an anime or anime-style image and has not named a specific image model. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. Set \"krea-2-turbo\" when the user asks for Krea 2 Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image and Krea 2 Turbo image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"/Qwen image models CANNOT do nudity. For permitted NSFW/nudity content, prefer \"dark-beast-krea2\", then \"dark-beast-z-turbo\"; \"chroma1-hd\", \"pony-v7\", \"chroma-detail\", \"chroma-v46-flash\", and \"z-turbo\" are compatible fallbacks."
|
|
64
68
|
},
|
|
65
69
|
"width": {
|
|
66
70
|
"type": "number",
|
|
67
|
-
"description": "Output image width in pixels. Default: 1024. Supported
|
|
71
|
+
"description": "Output image width in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"1080x1920\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
68
72
|
},
|
|
69
73
|
"height": {
|
|
70
74
|
"type": "number",
|
|
71
|
-
"description": "Output image height in pixels. Default: 1024. Supported
|
|
75
|
+
"description": "Output image height in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"1080x1920\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
72
76
|
},
|
|
73
77
|
"numberOfVariations": {
|
|
74
78
|
"type": "number",
|
|
75
|
-
"description": "Number of variations (1-16).
|
|
79
|
+
"description": "Number of variations (1-16). Set to the user's exact requested count in one call whenever they ask for multiple images and the outputs can share project settings. Trigger phrasings: \"draw N\", \"make N\", \"give me N\", \"show me N\", \"render N\", \"create N\", \"generate N\", \"N more\", \"another N\", \"N as separate\", \"N separate images\", \"N different images\", \"N options\", \"N takes\", \"N versions\", \"N variations\", \"N pictures of\", \"all at the same time\", \"in parallel\", \"side by side as separate\". This includes selection-gated image batches that will feed a later dance, animation, or video after the user picks one. Avoid multiple serial generate_image calls unless the user explicitly wants separate projects, isolated approvals, or per-output settings that cannot share one project. If the user previously got a composite \"N subjects in one image\" result and now says \"draw N more as separate images\" / \"as separate\" / \"separately\", set numberOfVariations=N for THIS call — the prior call's numberOfVariations does not carry forward when the user explicitly asks for separation. For screenplay/storyboard batches, this should equal the scene count and the prompt should contain one Dynamic Prompt branch with one full scene prompt per scene; do not set numberOfVariations=N with only one scene prompt. Default: 1 when the user clearly wants a single composite image (e.g. \"draw 2 goats in a meadow\" with no separation language, or explicit \"in one image\" / \"single image\" / \"composite\" / \"sheet\").",
|
|
76
80
|
"minimum": 1,
|
|
77
81
|
"maximum": 16
|
|
78
82
|
},
|
|
@@ -82,7 +86,7 @@
|
|
|
82
86
|
},
|
|
83
87
|
"starting_image_strength": {
|
|
84
88
|
"type": "number",
|
|
85
|
-
"description": "Image-to-image strength (0.0-1.0). Only
|
|
89
|
+
"description": "Image-to-image source guidance strength (0.0-1.0). Only set when a source image is available and the model supports img2img. For Z-Image/Z-Image Turbo, Krea 2 Turbo, or source-preserving enhancement requests, use 0.75 with sourceImageIndex so the source image remains a strong guide while allowing higher-resolution reconstruction. Use lower values only when the user explicitly asks for a lighter guide/subtle variation; higher values are more creative and can deviate further from the source."
|
|
86
90
|
},
|
|
87
91
|
"sourceImageIndex": {
|
|
88
92
|
"type": "number",
|
|
@@ -131,13 +135,13 @@
|
|
|
131
135
|
"type": "function",
|
|
132
136
|
"function": {
|
|
133
137
|
"name": "generate_video",
|
|
134
|
-
"description": "Generate a video from text or Seedance multimodal references. LTX 2.3 generates audio natively (dialogue, sounds, ambient music) — describe audio in the prompt. If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23
|
|
138
|
+
"description": "Generate a video from text or Seedance multimodal references. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively (dialogue, sounds, ambient music) — describe audio in the prompt. If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. For non-Seedance syncing to a specific song or audio track, use sound_to_video instead. For non-Seedance animation from a locked source photo, use animate_photo. Do NOT use for My Personas unless generating a Seedance reference-based video — standard persona videos use resolve_personas → edit_image → animate_photo. SEEDANCE DEFAULT: For seedance2, seedance2-mini, seedance2-fast, or seedance2-5, default to exactly one video (4-15s on 2.0/Mini/Fast, 4-30s on seedance2-5) unless the user explicitly asks for multiple separate outputs. Multiple beats, shots, or scene descriptions in one Seedance prompt within the selected model's per-clip limit are still one video. If the user requests one continuous Seedance video longer than 15s, prefer \"seedance2-5\", which renders up to 30s in a single call; beyond 30s (or on 2.0/Mini/Fast) preserve the requested total duration in the prompt/context and let chat orchestration split it into supported segment renders and stitch them instead of clamping it to a short excerpt. Uploaded/generated storyboard, shot-sheet, or trailer-concept images used as Seedance references should become one Seedance generate_video call by default; do not extract panels with edit_image and do not animate the storyboard sheet with LTX unless the user explicitly asks for separate non-Seedance clips. Seedance loose image, video, and audio references go through this tool; do not use animate_photo sourceImageIndex/frameRole/endImageIndex for Seedance. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. If the uploaded audio is the primary sync target, lip-sync target, or requested as sound-to-video/audio-sync, use sound_to_video with videoModel=\"seedance2-mini\" instead of this tool unless the user asks for full Seedance. Use referenceAudioIndices here only when audio is a loose reference under an image/video-anchored Seedance shot. For Seedance, every image — first frame, last frame, or loose reference — is passed through referenceImageIndices (auto-uploaded as referenceImageUrls). Anchor frame intent in the prompt with @Image tags such as \"Use @Image1 as the opening shot reference. Begin the video with a composition, subject placement, lighting, mood, and camera framing that closely match @Image1.\" (or @Image2 as the final shot reference). For seamless-loop or \"first frame and last frame identical\" requests with a single uploaded image, anchor it explicitly as both: \"Use @Image1 as both the first frame and last frame so the video loops cleanly back to the opening composition.\" Assign each useful @Image/@Video/@Audio tag a role. APPROVED STORYBOARD PRODUCTION: When the user asks for a production workflow from an approved storyboard, the chat orchestrator should use the durable CampaignStoryboard contract: render the composite board, audit it, generate per-scene GPT Image 2 keyframes, then render Seedance scene clips and stitch them. Do not replace that with a generic storyboard-reference video unless the user asks for a fast draft. PARTIAL VIDEO EDITS: Do NOT call generate_video to re-render an existing rendered/uploaded video just to change part of it (the bumper, the intro, the end card, a single scene, the last few seconds, etc.). Use replace_video_segment for that — it preserves the unchanged portion, keeps the original audio outside the replaced window, and costs far less. Likewise use extend_video to add new time to the end without rewriting the rest. If the request is vague, ask about vision/mood/style first. Only call once you have clear creative intent.",
|
|
135
139
|
"parameters": {
|
|
136
140
|
"type": "object",
|
|
137
141
|
"properties": {
|
|
138
142
|
"prompt": {
|
|
139
143
|
"type": "string",
|
|
140
|
-
"description": "Write one flowing paragraph like a cinematographer describing a shot. Present tense, specific natural language. Longer clips need longer prompts; close-ups need more detail than wide shots.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: shot/style → subject (age, clothing, hairstyle, distinguishing details) → environment, lighting, atmosphere → action beat by beat → camera movement → audio and dialogue.\n\nCAST CONTINUITY: For screenplay, script, storyboard, commercial, series, or other longer-form video tasks with recurring characters, use stable character names and repeat the same visual anchors every time they appear (age range, build, hairstyle, outfit silhouette, color palette, signature prop/accessory, posture, voice). Do not rename, merge, redesign, or drift characters between scenes unless the user asks.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Direct the layout like scene blocking. State left/right placement, foreground/background, facing toward/away, and relative distance when multiple subjects or important objects are involved.\n\nACTION: Drive motion with concrete verbs. Specify who moves, what moves, how it moves, and what the camera does. Avoid generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior — not \"she is sad\", instead \"she looks down, pauses, and her voice cracks\". QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nSTORYBOARD TEXT: For storyboard references, structural headings, section numbers, slide titles, panel titles, and captions may become short audio-only narration/voiceover or key-message beats, but they are not subtitles, title cards, lower thirds, or visible overlays unless the user explicitly asks for visible text/on-screen text/title card/subtitle/lower third/signage/CTA. Do not concatenate storyboard labels into run-on voiceover; use separate brief phrases with pauses.\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — close-up, tracking shot, dolly in, handheld, slow arc, static frame. Describe movement relative to subject.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters, dialogue, skits): capture the full creative intent. The system auto-expands into a detailed prompt.\n\nAVOID: Vague prompts, too many characters at once, conflicting lighting logic, readable text or logos, abstract emotions with no visible behavior, rigid numeric constraints (exact angles, counts, speeds).\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax. Lock in any camera/subject/style the user specified, vary the rest. Example: \"slow dolly in on a city street {at dawn with golden light|during a rainstorm|at night with neon reflections}\"."
|
|
144
|
+
"description": "Write one flowing paragraph like a cinematographer describing a shot. Present tense, specific natural language. Longer clips need longer prompts; close-ups need more detail than wide shots.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: shot/style → subject (age, clothing, hairstyle, distinguishing details) → environment, lighting, atmosphere → action beat by beat → camera movement → audio and dialogue.\n\nCAST CONTINUITY: For screenplay, script, storyboard, commercial, series, or other longer-form video tasks with recurring characters, use stable character names and repeat the same visual anchors every time they appear (age range, build, hairstyle, outfit silhouette, color palette, signature prop/accessory, posture, voice). Do not rename, merge, redesign, or drift characters between scenes unless the user asks.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Direct the layout like scene blocking. State left/right placement, foreground/background, facing toward/away, and relative distance when multiple subjects or important objects are involved.\n\nACTION: Drive motion with concrete verbs. Specify who moves, what moves, how it moves, and what the camera does. Avoid generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior — not \"she is sad\", instead \"she looks down, pauses, and her voice cracks\". QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nSTORYBOARD TEXT: For storyboard references, structural headings, section numbers, slide titles, panel titles, and captions may become short audio-only narration/voiceover or key-message beats, but they are not subtitles, title cards, lower thirds, or visible overlays unless the user explicitly asks for visible text/on-screen text/title card/subtitle/lower third/signage/CTA. Do not concatenate storyboard labels into run-on voiceover; use separate brief phrases with pauses.\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — close-up, tracking shot, dolly in, handheld, slow arc, static frame. Describe movement relative to subject.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters, dialogue, skits): capture the full creative intent. The system auto-expands into a detailed prompt.\n\nAVOID: Vague prompts, too many characters at once, conflicting lighting logic, readable text or logos, abstract emotions with no visible behavior, rigid numeric constraints (exact angles, counts, speeds).\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same references, model, duration, dimensions, and generation parameters and only prompt text varies. Lock in any camera/subject/style the user specified, vary the rest. Example: \"slow dolly in on a city street {at dawn with golden light|during a rainstorm|at night with neon reflections}\"."
|
|
141
145
|
},
|
|
142
146
|
"expandPrompt": {
|
|
143
147
|
"type": "boolean",
|
|
@@ -149,64 +153,74 @@
|
|
|
149
153
|
},
|
|
150
154
|
"duration": {
|
|
151
155
|
"type": "number",
|
|
152
|
-
"description": "Video duration in seconds. Default: 5. Range: 2-20. Use when the user explicitly requests a specific length.",
|
|
156
|
+
"description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.5, LTX 2.3, and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30 — only \"seedance2-5\" can use the 16-30s part of this range. MiniMax H3 is quantized to a 17-frame grid at a fixed 24 fps and renders 124-362 frames, so an H3 clip runs 5.17-15.08 seconds and a requested length outside that window snaps to the nearest valid H3 length. Use when the user explicitly requests a specific length.",
|
|
153
157
|
"minimum": 2,
|
|
154
|
-
"maximum":
|
|
158
|
+
"maximum": 30
|
|
155
159
|
},
|
|
156
160
|
"negativePrompt": {
|
|
157
161
|
"type": "string",
|
|
158
|
-
"description": "
|
|
162
|
+
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. All standard LTX 2.5 workflow IDs accept this separate negative prompt. Use this field only when the user explicitly asks to set one. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt. Do not set for MiniMax H3, Seedance, or HappyHorse."
|
|
159
163
|
},
|
|
160
164
|
"videoModel": {
|
|
161
165
|
"type": "string",
|
|
162
166
|
"enum": [
|
|
167
|
+
"ltx25",
|
|
163
168
|
"ltx23",
|
|
164
169
|
"wan22",
|
|
165
170
|
"seedance2",
|
|
166
|
-
"seedance2-
|
|
171
|
+
"seedance2-mini",
|
|
172
|
+
"seedance2-fast",
|
|
173
|
+
"seedance2-5",
|
|
174
|
+
"minimax-h3-t2v",
|
|
175
|
+
"minimax-h3-t2v-turbo",
|
|
176
|
+
"happyhorse-1.1-t2v",
|
|
177
|
+
"happyhorse-1.1-i2v",
|
|
178
|
+
"happyhorse-1.1-r2v",
|
|
179
|
+
"minimax-h3-r2v",
|
|
180
|
+
"minimax-h3-r2v-turbo"
|
|
167
181
|
],
|
|
168
|
-
"description": "Video model. \"
|
|
182
|
+
"description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" only when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft, Mini, or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0, Mini, and Fast accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with no more than 30 reference media files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
|
|
169
183
|
},
|
|
170
184
|
"generateAudio": {
|
|
171
185
|
"type": "boolean",
|
|
172
|
-
"description": "
|
|
186
|
+
"description": "Whether to include generated/native audio for audio-capable models. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. When false, the returned video has no audio track. Not supported by WAN or HappyHorse."
|
|
173
187
|
},
|
|
174
188
|
"referenceImageIndices": {
|
|
175
189
|
"type": "array",
|
|
176
190
|
"items": {
|
|
177
191
|
"type": "number"
|
|
178
192
|
},
|
|
179
|
-
"description": "
|
|
193
|
+
"description": "Image references for Seedance (@Image tags), HappyHorse 1.1 r2v, and MiniMax H3 r2v. Use negative indices for uploaded images (-1 first upload, -2 second upload) and non-negative indices for generated image results. For Seedance, omit by default: uploaded images are auto-forwarded as @Image references. Anchor frame intent in the prompt with @Image tags: \"Use @Image1 as the opening shot reference. Begin the video with a composition, subject placement, lighting, mood, and camera framing that closely match @Image1.\" (or @Image2 as the final shot reference). For seamless-loop or \"first frame and last frame identical\" requests with a single uploaded image, anchor it explicitly as both: \"Use @Image1 as both the first frame and last frame so the video loops cleanly back to the opening composition.\" Do not use animate_photo sourceImageIndex/frameRole/endImageIndex for Seedance. For HappyHorse 1.1 r2v, pass 1-9 image references. For MiniMax H3 r2v, up to 9 images are accepted; at least one image or video reference is required, and H3 references are loose references, not locked frames, and are addressed in the prompt as <Picture 1>, <Picture 2>, and so on in selection order."
|
|
180
194
|
},
|
|
181
195
|
"referenceVideoIndices": {
|
|
182
196
|
"type": "array",
|
|
183
197
|
"items": {
|
|
184
198
|
"type": "number"
|
|
185
199
|
},
|
|
186
|
-
"description": "
|
|
200
|
+
"description": "Optional loose video references for Seedance and MiniMax H3 r2v. Use negative indices for uploaded videos (-1 first uploaded video, -2 second uploaded video) and non-negative indices for generated video results. For Seedance, omit by default: uploaded videos are auto-forwarded as @Video references. Set to choose a subset or include previously generated video URLs. Do not use this for uploaded source-video transforms, upscales, enhancements, restyles, or remasters; use video_to_video with controlMode=\"seedance-v2v\" instead. For MiniMax H3 r2v, up to 3 reference videos (24 fps, 2-15s each, optional soundtrack), addressed as <Video 1>, <Video 2>, and so on in selection order; they can satisfy the required visual reference without an image."
|
|
187
201
|
},
|
|
188
202
|
"referenceAudioIndices": {
|
|
189
203
|
"type": "array",
|
|
190
204
|
"items": {
|
|
191
205
|
"type": "number"
|
|
192
206
|
},
|
|
193
|
-
"description": "
|
|
207
|
+
"description": "Optional loose audio references for Seedance and MiniMax H3 r2v. Use negative indices for uploaded audio files (-1 first uploaded audio, -2 second uploaded audio) and non-negative indices for generated audio results. For Seedance, omit by default: uploaded audio is auto-forwarded as @Audio references when the Seedance request also has an image or video reference. Use this only for loose background, mood, timing, or style references under an image/video-anchored Seedance shot. If the uploaded audio is the primary sync target, lip-sync target, or requested as sound-to-video/audio-sync, use sound_to_video with videoModel=\"seedance2-mini\" instead unless the user asks for full Seedance. Audio-only Seedance requests are unsupported; use sound_to_video for uploaded-audio-only workflows. For MiniMax H3 r2v, up to 3 standalone audio tracks, addressed as <Audio 1>, <Audio 2>, and so on in selection order — a reference video's own soundtrack takes its Audio number before standalone tracks; they supplement a required image or video reference and cannot be the sole input."
|
|
194
208
|
},
|
|
195
209
|
"width": {
|
|
196
210
|
"type": "number",
|
|
197
|
-
"description": "Video width in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Default resolution depends on model and quality tier: LTX Fast about 720p and High/Pro about 1080p; WAN Fast uses 480p short side and High/Pro uses 720p short side. Set width only when the user specifies an exact width or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target, not an instruction to make landscape 1280x720. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override the default media quality. Mappings when orientation is explicit: 480p landscape=854x480, 480p portrait=480x854, 720p landscape=1280x720, 720p portrait=720x1280, 1080p landscape=1920x1080, 1080p portrait=1080x1920, 4K landscape=3840x2160. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
211
|
+
"description": "Video width in pixels. LTX 2.5 and LTX 2.3: 640-3840. WAN: 480-1536. Default resolution depends on model and quality tier: LTX Fast about 720p and High/Pro about 1080p; WAN Fast uses 480p short side and High/Pro uses 720p short side. Set width only when the user specifies an exact width or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target, not an instruction to make landscape 1280x720. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override the default media quality. Mappings when orientation is explicit: 480p landscape=854x480, 480p portrait=480x854, 720p landscape=1280x720, 720p portrait=720x1280, 1080p landscape=1920x1080, 1080p portrait=1080x1920, 4K landscape=3840x2160. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
198
212
|
},
|
|
199
213
|
"height": {
|
|
200
214
|
"type": "number",
|
|
201
|
-
"description": "Video height in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Set height only when the user specifies an exact height or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target; do not convert it to landscape dimensions unless the user says landscape/horizontal/widescreen. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override Default Media Quality, including Pro. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
215
|
+
"description": "Video height in pixels. LTX 2.5 and LTX 2.3: 640-3840. WAN: 480-1536. Set height only when the user specifies an exact height or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target; do not convert it to landscape dimensions unless the user says landscape/horizontal/widescreen. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override Default Media Quality, including Pro. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
202
216
|
},
|
|
203
217
|
"targetResolution": {
|
|
204
218
|
"type": "number",
|
|
205
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
219
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-mini\" vs \"seedance2-fast\" vs \"seedance2-5\"). Seedance 2.0 full supports 4K; Seedance Mini, Fast, and Seedance 2.5 support 480p/720p only, so never set 1080p or 4K for \"seedance2-5\". Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
|
|
206
220
|
},
|
|
207
221
|
"numberOfVariations": {
|
|
208
222
|
"type": "number",
|
|
209
|
-
"description": "Number of variations (1-16). Use 1 unless user explicitly requests multiple separate video outputs. For Seedance, default to 1
|
|
223
|
+
"description": "Number of variations (1-16). Use with one Dynamic Prompt branch for multiple prompt-only takes that share the same references, model, duration, dimensions, and parameters. This creates one Sogni project with multiple jobs. Use 1 unless the user explicitly requests multiple separate video outputs. For Seedance, default to 1 unless the user explicitly requests separate outputs.",
|
|
210
224
|
"minimum": 1,
|
|
211
225
|
"maximum": 16
|
|
212
226
|
},
|
|
@@ -216,7 +230,7 @@
|
|
|
216
230
|
},
|
|
217
231
|
"voicePersonaName": {
|
|
218
232
|
"type": "string",
|
|
219
|
-
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one described in the video (e.g. \"David\" narrates a scene featuring Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23."
|
|
233
|
+
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one described in the video (e.g. \"David\" narrates a scene featuring Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
|
|
220
234
|
}
|
|
221
235
|
},
|
|
222
236
|
"required": [
|
|
@@ -235,7 +249,7 @@
|
|
|
235
249
|
"properties": {
|
|
236
250
|
"prompt": {
|
|
237
251
|
"type": "string",
|
|
238
|
-
"description": "Genre, mood, and style description for the music. Be specific about musical characteristics.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing.\n\nExamples:\n- \"upbeat electronic dance music with driving bass and synth arpeggios\"\n- \"mellow jazz ballad with soft piano, brushed drums, and walking bass\"\n- \"epic orchestral soundtrack with soaring strings and powerful brass\"\n- \"lo-fi hip hop beat with vinyl crackle, muted keys, and chill vibes\"\n- \"acoustic folk song with fingerpicked guitar and warm harmonies\"\n\nInclude:\n- Genre (rock, jazz, electronic, classical, hip-hop, etc.)\n- Mood (happy, melancholic, energetic, relaxing, epic, etc.)\n- Instruments (piano, guitar, drums, synth, strings, etc.)\n- Style descriptors (driving, mellow, atmospheric, punchy, etc.)\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary ONE dimension across separate tracks. Lock in any genre/mood/instruments the user specified, vary the rest. Example: \"{lo-fi hip hop beat with muted keys|jazz piano trio with brushed drums|ambient electronic with soft pads} with warm reverb and vinyl texture\"."
|
|
252
|
+
"description": "Genre, mood, and style description for the music. Be specific about musical characteristics.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing.\n\nExamples:\n- \"upbeat electronic dance music with driving bass and synth arpeggios\"\n- \"mellow jazz ballad with soft piano, brushed drums, and walking bass\"\n- \"epic orchestral soundtrack with soaring strings and powerful brass\"\n- \"lo-fi hip hop beat with vinyl crackle, muted keys, and chill vibes\"\n- \"acoustic folk song with fingerpicked guitar and warm harmonies\"\n\nInclude:\n- Genre (rock, jazz, electronic, classical, hip-hop, etc.)\n- Mood (happy, melancholic, energetic, relaxing, epic, etc.)\n- Instruments (piano, guitar, drums, synth, strings, etc.)\n- Style descriptors (driving, mellow, atmospheric, punchy, etc.)\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary ONE dimension across separate tracks. This is one Sogni project with multiple jobs, so prefer it when all tracks share the same duration, BPM, key, lyrics, model, and generation parameters and only prompt text varies. Lock in any genre/mood/instruments the user specified, vary the rest. Example: \"{lo-fi hip hop beat with muted keys|jazz piano trio with brushed drums|ambient electronic with soft pads} with warm reverb and vinyl texture\"."
|
|
239
253
|
},
|
|
240
254
|
"duration": {
|
|
241
255
|
"type": "number",
|
|
@@ -261,9 +275,10 @@
|
|
|
261
275
|
"type": "string",
|
|
262
276
|
"enum": [
|
|
263
277
|
"turbo",
|
|
264
|
-
"sft"
|
|
278
|
+
"sft",
|
|
279
|
+
"music3"
|
|
265
280
|
],
|
|
266
|
-
"description": "
|
|
281
|
+
"description": "Music model. \"turbo\" (default): ACE-Step 1.5 Turbo — fast 4-16 step drafts at half cost. \"sft\": ACE-Step 1.5 SFT — experimental, strong lyric handling, 10-200 steps, full cost. \"music3\": MiniMax Music 3 — premium autoregressive composer with the best vocals, lyric adherence and song structure; 30 steps, up to 5 minutes, ~20x turbo cost, and it treats duration as a ceiling (may end the song early at a musical resolution). Use music3 when the user asks for the best quality, realistic vocals, or full songs; otherwise default to \"turbo\"."
|
|
267
282
|
},
|
|
268
283
|
"timesig": {
|
|
269
284
|
"type": "number",
|
|
@@ -277,7 +292,7 @@
|
|
|
277
292
|
},
|
|
278
293
|
"numberOfVariations": {
|
|
279
294
|
"type": "number",
|
|
280
|
-
"description": "Number of variations (1-16). Use
|
|
295
|
+
"description": "Number of variations (1-16). Use with one Dynamic Prompt branch when the user requests multiple prompt-only music variations that share the same duration, BPM, key, lyrics, model, and parameters. This creates one Sogni project with multiple jobs. Default: 1.",
|
|
281
296
|
"minimum": 1,
|
|
282
297
|
"maximum": 16
|
|
283
298
|
}
|
|
@@ -292,13 +307,13 @@
|
|
|
292
307
|
"type": "function",
|
|
293
308
|
"function": {
|
|
294
309
|
"name": "edit_image",
|
|
295
|
-
"description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images,
|
|
310
|
+
"description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images, Qwen up to 3 images, and Krea 2 Identity Edit / Dark Beast Krea 2 Identity Edit up to 2 images. Best for style-guided generation, combining elements from multiple images, ANY persona image creation, identity-preserving Krea edits, and any uploaded brand asset reuse — logos, brand marks, mascots, product shots, photos, screenshots, sketches, or character designs the user expects to appear in or guide the result. ALWAYS use this (never generate_image) when persona photos OR uploaded image assets meant for reuse are in context — even if a specific edit model is requested. Exception: explicit Z-image/Z-image Turbo/Krea 2 Turbo uploaded-image enhancement uses generate_image with sourceImageIndex and starting_image_strength because those base image-to-image models are not edit_image models. If a previous edit_image attempt did not preserve the uploaded asset well, stay on edit_image and tighten the prompt or switch model; generate_image has no access to the upload.",
|
|
296
311
|
"parameters": {
|
|
297
312
|
"type": "object",
|
|
298
313
|
"properties": {
|
|
299
314
|
"prompt": {
|
|
300
315
|
"type": "string",
|
|
301
|
-
"description": "Edit instruction describing what to generate using the reference images as guidance. 50-200 words recommended.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing.\n\nPROMPT CONSTRUCTION ORDER — build the prompt in this sequence:\n1. IDENTITY LOCK — state which picture owns the person's identity (GOLDEN RULE: never leave identity ambiguous when editing a person)\n2. REQUESTED EDIT — describe only what CHANGES (the delta), not the whole image\n3. REFERENCE ROLE MAPPING — assign each picture ONE primary role: base_identity (face/person), pose_reference, outfit_reference, style_reference, background_reference, or color_reference\n4. POSE / COMPOSITION — pose, framing, camera angle (omit if unchanged)\n5. STYLE — artistic style, genre, era (omit if unchanged)\n6. LIGHTING / REALISM — \"maintain realistic anatomy, perspective, and lighting integration\"\n7. PRESERVE clause — always end with \"preserve all unmentioned details\"\n\nIDENTITY LOCK (required when a person is in any reference image):\n\"Preserve the exact facial likeness from picture N — face structure, eye shape, nose shape, mouth shape, jawline, skin tone, hairline, apparent age, and overall recognizability.\"\nNever let a style, pose, or clothing reference silently override the face. If multiple images are provided, explicitly state \"identity comes only from picture N — do not borrow identity from other pictures.\"\n\nMINIMAL-CHANGE PRINCIPLE: The base image already contains the subject, composition, camera angle, expression, lighting, and background. Describe only the delta. Use positive constraints (\"preserve exact facial likeness\") not negative ones (\"don't change the face\").\n\nSINGLE-IMAGE PATTERN:\n\"Preserve the exact facial likeness and recognizability of the person from picture 1. [Describe only the requested change]. Keep the same pose, framing, camera angle, and expression unless the user specifically requests changes to these. Preserve all unmentioned details.\"\n\nMULTI-IMAGE PATTERN:\n\"Use the person from picture 1 as the final subject and preserve their exact facial likeness. [Requested edit]. Identity comes only from picture 1. Pose from picture 2. Outfit from picture 3. Do not borrow identity from pictures 2 or 3. Maintain realistic anatomy, perspective, and lighting integration. Preserve all unmentioned details.\"\n\nCREATIVE TRANSFORMATIONS — be vivid and reference-specific, name the artist, franchise, or era, but always anchor identity first:\n - \"Preserve the exact facial likeness from picture 1. Transform them into a Renaissance oil painting in the style of Vermeer — rich warm tones, dramatic chiaroscuro lighting, ornate period clothing. Maintain realistic anatomy. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Reimagine them as a Marvel superhero — cinematic dramatic lighting, heroic pose, detailed costume with cape, glowing energy effects. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Transform them into a Studio Ghibli anime character — soft watercolor backgrounds, gentle Ghibli-style rendering, whimsical atmosphere. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Place them into a Star Wars scene — Jedi robes, lightsaber glow, dramatic sci-fi backdrop. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Turn them into a GTA loading screen character — bold outlines, saturated colors, attitude-filled pose, urban backdrop. Preserve all unmentioned details.\"\n\nFAILURE MODES TO AVOID:\n- Face drift: identity source not specified, or style/pose reference overrides the face\n- Over-editing: for simple edits, prompt rewrites the entire image instead of describing the delta (creative transformations may intentionally change more)\n- Reference confusion: multiple images provided without explicit role mapping\n\nCHARACTER / MASCOT SHEETS: When the user asks for a character sheet, mascot sheet, model sheet, turnaround, expression sheet, or reusable character reference board using uploaded references, create ONE comprehensive professional reference-board image, not separate variations. Map reference roles clearly first (for example: picture 1 = character identity/style reference, picture 2 = logo/brand asset) and keep the character identity consistent across every panel. Include a large hero pose, front / 3/4 / side / back turnaround views, an expression row, action/personality poses, accessories or props, color palette swatches, and compact notes such as personality, fun facts, or brand usage when appropriate. Preserve exact user-provided brand names, slogans, logo text, and requested copy verbatim; incidental tiny notes may be generated by the image model if the user did not provide exact wording. Use clean readable typography.\n\nBATCH VARIATIONS: When numberOfVariations > 1, the prompt must describe ONE subject in ONE scene — never mention counts, \"versions\", \"different\", or \"multiple\" in the prompt text. NEVER describe multiple copies or duplicates of the subject in a single image (no grids, collages, or side-by-side). Use Dynamic Prompt syntax to vary ONE dimension across separate images. For personas: vary scene, activity, expression, or environment — never vary identity. Example: user asks \"4 versions at the beach\" → numberOfVariations=4, prompt=\"[persona] at the beach {building a sandcastle|surfing a wave|reading under a palm tree|flying a kite}\" — each output is ONE person doing ONE activity. For direct edits: vary the approach, e.g., numberOfVariations=3, prompt=\"make the sky {a vibrant sunset|stormy and dramatic|clear blue}\". Preserve any requested orientation, aspect ratio, or exact pixel dimensions across every variation.\n\nSELECTION-GATED IMAGE STAGES: If the user asks for multiple reference-guided image options/takes/versions and says they will pick one before a later dance, animation, or video, this edit_image call is still the first step. Generate the complete image batch now with sourceImageIndex set to the relevant reference, the exact requested count, Dynamic Prompt options for each output, and the final video/image aspect ratio. Do not ask the user to choose before the images exist, and do not call video tools until after the user selects an image.\n\nLINKED VARIANTS: If multiple details must stay paired per output — visual style, identity cues, outfit, label text, symbols, setting, character, prop, location, or before/after keyframe details — use ONE top-level Dynamic Prompt branch with one complete prompt per output. Do NOT use separate Dynamic Prompt groups for details that must stay together; unpaired groups can mix attributes. If the user asks for per-variant facial, identity, or appearance changes, repeat that guidance inside EVERY option while also preserving recognizability. When the user names a subject or character, write that name or stable role inside every Dynamic Prompt option; a shared prefix outside the branch is not enough because each option must stand alone as a complete identity contract.\n\nEach option must be a fully concrete description — name the actual garment or styling, the actual setting, the actual accessories, and the literal text or symbol shown on screen when requested. Never use meta-placeholder phrasing such as \"style-specific outfit\", \"variant-specific background\", \"include the requested symbol\", \"include a humorous alternate name\", or \"bake the name and symbol into the image\" — those describe the task instead of the image.\n\nORIGINAL + VARIANT BATCHES: When one option is a remade/preserved original and the other options are themed variants, the original option still needs a concrete visual contract. Say to preserve the original clothing/wardrobe/outfit and original background/setting, then name any requested added text, label, flag, logo, symbol, or prop for that original option. Do not leave the original option as only \"unmodified original person\"; it must be as fully specified as every themed option.\n\nNEW SETTING PER OPTION: When the variant theme implies a new place, culture, era, or context, every option must name its own setting (location, props, lighting). Do NOT carry the source background forward, do NOT write \"in the same pose and placement as the original photo\" without also naming the new background, and do NOT rely on \"preserve all unmentioned details\" to handle the setting — the new setting IS a mentioned detail.\n\nRECOGNIZABILITY OVER FEATURE LOCK: For ethnic / age / character / art-style transformations, do NOT paste the strict IDENTITY LOCK feature list (\"face structure, eye shape, nose shape, mouth shape, jawline, skin tone, hairline\") inside each option — that list contradicts the requested face change and the source face will pass through unchanged. Anchor recognizability per option through apparent age, signature hair silhouette, build, posture, and expression, and explicitly allow skin tone, facial features, and proportions to shift toward the target.\n\nCorrect shape (each option self-contained, concrete, with a fresh setting and a recognizability anchor instead of a strict feature lock):\n\"{The subject wearing [specific garment, color, cut, and material], standing in [specific NEW setting with props and lighting — never the source background], bold text at the bottom reads [literal requested text], [specific requested visual symbol] appears as a sign or prop, [requested per-variant facial or appearance shift, e.g. \"skin tone, eye shape, and bone structure shift toward <target> features\"], recognizable through apparent age, signature hair silhouette, build, posture, and expression|The subject wearing [second specific garment, color, cut, and material], standing in [second specific NEW setting with props and lighting], bold text at the bottom reads [second literal requested text], [second requested visual symbol] appears as a sign or prop, [second requested facial or appearance shift], recognizable through apparent age, signature hair silhouette, build, posture, and expression|...}\"\n\nWrong shape (placeholder labels masquerading as prompts):\n\"{First variant with variant-specific facial features, placeholder wardrobe, alternate name, and requested symbol baked in|Second variant with different variant-specific facial features, placeholder wardrobe, alternate name, and requested symbol baked in|...}\"\n\nAlso wrong (strict feature lock + no new setting — the source face and source background pass through unchanged):\n\"{Preserve the exact facial likeness — face structure, eye shape, nose shape, mouth shape, jawline, skin tone, hairline. Reimagine as <variant>: [garment description], standing in the exact same pose and placement as the original photo. Preserve all unmentioned details.|Preserve the exact facial likeness — [same strict lock]. Reimagine as <other variant>: [other garment], standing in the exact same pose and placement as the original photo. Preserve all unmentioned details.|...}\"\n\nSCREENPLAY / STORYBOARD BATCHES: For multi-scene story, commercial, or longer-form video keyframes, use one Dynamic Prompt branch with one full scene prompt per option. Recurring characters must keep stable names and repeated visual anchors in every scene option where they appear: face/identity source if available, age range, build, hairstyle, outfit silhouette, color palette, signature prop/accessory, posture, and role. Do not let style, scene changes, or pose references alter identity. Include screenplay-style speaker tags when dialogue matters, e.g. CHARACTER: \"We made it.\"\n\nCOMPOSITE GPT IMAGE 2 STORYBOARD SHEETS: When numberOfVariations=1 and the user asks for one composite video storyboard/keyframe sheet using uploaded or generated references, the prompt must be a compiled storyboard prompt, not a concept summary. Include a SCENES: section with exactly the requested number of concrete entries named SCENE_01, SCENE_02, etc. Every scene entry must include Visual/Action, Camera/Motion, Dialogue/VO (or [no dialogue]), Audio/SFX, and any visible text or reference usage for that scene. Do not provide only the source brief or generic layout instructions; malformed compiled storyboard prompts are blocked by quality audit."
|
|
316
|
+
"description": "Edit instruction describing what to generate using the reference images as guidance. 50-200 words recommended.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing.\n\nPROMPT CONSTRUCTION ORDER — build the prompt in this sequence:\n1. IDENTITY LOCK — state which picture owns the person's identity (GOLDEN RULE: never leave identity ambiguous when editing a person)\n2. REQUESTED EDIT — describe only what CHANGES (the delta), not the whole image\n3. REFERENCE ROLE MAPPING — assign each picture ONE primary role: base_identity (face/person), pose_reference, outfit_reference, style_reference, background_reference, or color_reference\n4. POSE / COMPOSITION — pose, framing, camera angle (omit if unchanged)\n5. STYLE — artistic style, genre, era (omit if unchanged)\n6. LIGHTING / REALISM — \"maintain realistic anatomy, perspective, and lighting integration\"\n7. PRESERVE clause — always end with \"preserve all unmentioned details\"\n\nIDENTITY LOCK (required when a person is in any reference image):\n\"Preserve the exact facial likeness from picture N — face structure, eye shape, nose shape, mouth shape, jawline, skin tone, hairline, apparent age, and overall recognizability.\"\nNever let a style, pose, or clothing reference silently override the face. If multiple images are provided, explicitly state \"identity comes only from picture N — do not borrow identity from other pictures.\"\n\nMINIMAL-CHANGE PRINCIPLE: The base image already contains the subject, composition, camera angle, expression, lighting, and background. Describe only the delta. Use positive constraints (\"preserve exact facial likeness\") not negative ones (\"don't change the face\").\n\nSINGLE-IMAGE PATTERN:\n\"Preserve the exact facial likeness and recognizability of the person from picture 1. [Describe only the requested change]. Keep the same pose, framing, camera angle, and expression unless the user specifically requests changes to these. Preserve all unmentioned details.\"\n\nMULTI-IMAGE PATTERN:\n\"Use the person from picture 1 as the final subject and preserve their exact facial likeness. [Requested edit]. Identity comes only from picture 1. Pose from picture 2. Outfit from picture 3. Do not borrow identity from pictures 2 or 3. Maintain realistic anatomy, perspective, and lighting integration. Preserve all unmentioned details.\"\n\nCREATIVE TRANSFORMATIONS — be vivid and reference-specific, name the artist, franchise, or era, but always anchor identity first:\n - \"Preserve the exact facial likeness from picture 1. Transform them into a Renaissance oil painting in the style of Vermeer — rich warm tones, dramatic chiaroscuro lighting, ornate period clothing. Maintain realistic anatomy. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Reimagine them as a Marvel superhero — cinematic dramatic lighting, heroic pose, detailed costume with cape, glowing energy effects. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Transform them into a Studio Ghibli anime character — soft watercolor backgrounds, gentle Ghibli-style rendering, whimsical atmosphere. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Place them into a Star Wars scene — Jedi robes, lightsaber glow, dramatic sci-fi backdrop. Preserve all unmentioned details.\"\n - \"Preserve the exact facial likeness from picture 1. Turn them into a GTA loading screen character — bold outlines, saturated colors, attitude-filled pose, urban backdrop. Preserve all unmentioned details.\"\n\nFAILURE MODES TO AVOID:\n- Face drift: identity source not specified, or style/pose reference overrides the face\n- Over-editing: for simple edits, prompt rewrites the entire image instead of describing the delta (creative transformations may intentionally change more)\n- Reference confusion: multiple images provided without explicit role mapping\n\nCHARACTER / MASCOT SHEETS: When the user asks for a character sheet, mascot sheet, model sheet, turnaround, expression sheet, or reusable character reference board using uploaded references, create ONE comprehensive professional reference-board image, not separate variations. Map reference roles clearly first (for example: picture 1 = character identity/style reference, picture 2 = logo/brand asset) and keep the character identity consistent across every panel. Include a large hero pose, front / 3/4 / side / back turnaround views, an expression row, action/personality poses, accessories or props, color palette swatches, and compact notes such as personality, fun facts, or brand usage when appropriate. Preserve exact user-provided brand names, slogans, logo text, and requested copy verbatim; incidental tiny notes may be generated by the image model if the user did not provide exact wording. Use clean readable typography.\n\nBATCH VARIATIONS: When numberOfVariations > 1, the prompt describes one output image. Do not mention counts, \"versions\", \"different\", or \"multiple\" in the prompt text unless the user explicitly wants those words visible in the image. Do not describe multiple copies or duplicates of the subject in a single image unless the user asked for a grid, collage, or side-by-side composition. Use Dynamic Prompt syntax to vary one dimension across separate images. For personas: vary scene, activity, expression, or environment; preserve identity. Example: user asks \"4 versions at the beach\" → numberOfVariations=4, prompt=\"[persona] at the beach {building a sandcastle|surfing a wave|reading under a palm tree|flying a kite}\" — each output is one person doing one activity. For direct edits: vary the approach, e.g., numberOfVariations=3, prompt=\"make the sky {a vibrant sunset|stormy and dramatic|clear blue}\". Preserve any requested orientation, aspect ratio, or exact pixel dimensions across every variation.\n\nSELECTION-GATED IMAGE STAGES: If the user asks for multiple reference-guided image options/takes/versions and says they will pick one before a later dance, animation, or video, this edit_image call is still the first step. Generate the complete image batch now with sourceImageIndex set to the relevant reference, the exact requested count, Dynamic Prompt options for each output, and the final video/image aspect ratio. Do not ask the user to choose before the images exist, and do not call video tools until after the user selects an image.\n\nLINKED VARIANTS: If multiple details must stay paired per output — visual style, identity cues, outfit, label text, symbols, setting, character, prop, location, or before/after keyframe details — use ONE top-level Dynamic Prompt branch with one complete prompt per output. Do NOT use separate Dynamic Prompt groups for details that must stay together; unpaired groups can mix attributes. If the user asks for per-variant facial, identity, or appearance changes, repeat that guidance inside EVERY option while also preserving recognizability. When the user names a subject or character, write that name or stable role inside every Dynamic Prompt option; a shared prefix outside the branch is not enough because each option must stand alone as a complete identity contract.\n\nEach option must be a fully concrete description — name the actual garment or styling, the actual setting, the actual accessories, and the literal text or symbol shown on screen when requested. Never use meta-placeholder phrasing such as \"style-specific outfit\", \"variant-specific background\", \"include the requested symbol\", \"include a humorous alternate name\", or \"bake the name and symbol into the image\" — those describe the task instead of the image.\n\nORIGINAL + VARIANT BATCHES: When one option is a remade/preserved original and the other options are themed variants, the original option still needs a concrete visual contract. Say to preserve the original clothing/wardrobe/outfit and original background/setting, then name any requested added text, label, flag, logo, symbol, or prop for that original option. Do not leave the original option as only \"unmodified original person\"; it must be as fully specified as every themed option.\n\nNEW SETTING PER OPTION: When the variant theme implies a new place, culture, era, or context, every option must name its own setting (location, props, lighting). Do NOT carry the source background forward, do NOT write \"in the same pose and placement as the original photo\" without also naming the new background, and do NOT rely on \"preserve all unmentioned details\" to handle the setting — the new setting IS a mentioned detail.\n\nRECOGNIZABILITY OVER FEATURE LOCK: For ethnic / age / character / art-style transformations, do NOT paste the strict IDENTITY LOCK feature list (\"face structure, eye shape, nose shape, mouth shape, jawline, skin tone, hairline\") inside each option — that list contradicts the requested face change and the source face will pass through unchanged. Anchor recognizability per option through apparent age, signature hair silhouette, build, posture, and expression, and explicitly allow skin tone, facial features, and proportions to shift toward the target.\n\nCorrect shape (each option self-contained, concrete, with a fresh setting and a recognizability anchor instead of a strict feature lock):\n\"{The subject wearing [specific garment, color, cut, and material], standing in [specific NEW setting with props and lighting — never the source background], bold text at the bottom reads [literal requested text], [specific requested visual symbol] appears as a sign or prop, [requested per-variant facial or appearance shift, e.g. \"skin tone, eye shape, and bone structure shift toward <target> features\"], recognizable through apparent age, signature hair silhouette, build, posture, and expression|The subject wearing [second specific garment, color, cut, and material], standing in [second specific NEW setting with props and lighting], bold text at the bottom reads [second literal requested text], [second requested visual symbol] appears as a sign or prop, [second requested facial or appearance shift], recognizable through apparent age, signature hair silhouette, build, posture, and expression|...}\"\n\nWrong shape (placeholder labels masquerading as prompts):\n\"{First variant with variant-specific facial features, placeholder wardrobe, alternate name, and requested symbol baked in|Second variant with different variant-specific facial features, placeholder wardrobe, alternate name, and requested symbol baked in|...}\"\n\nAlso wrong (strict feature lock + no new setting — the source face and source background pass through unchanged):\n\"{Preserve the exact facial likeness — face structure, eye shape, nose shape, mouth shape, jawline, skin tone, hairline. Reimagine as <variant>: [garment description], standing in the exact same pose and placement as the original photo. Preserve all unmentioned details.|Preserve the exact facial likeness — [same strict lock]. Reimagine as <other variant>: [other garment], standing in the exact same pose and placement as the original photo. Preserve all unmentioned details.|...}\"\n\nSCREENPLAY / STORYBOARD BATCHES: For multi-scene story, commercial, or longer-form video keyframes, use one Dynamic Prompt branch with one full scene prompt per option. Recurring characters must keep stable names and repeated visual anchors in every scene option where they appear: face/identity source if available, age range, build, hairstyle, outfit silhouette, color palette, signature prop/accessory, posture, and role. Do not let style, scene changes, or pose references alter identity. Include screenplay-style speaker tags when dialogue matters, e.g. CHARACTER: \"We made it.\"\n\nCOMPOSITE GPT IMAGE 2 STORYBOARD SHEETS: When numberOfVariations=1 and the user asks for one composite video storyboard/keyframe sheet using uploaded or generated references, the prompt must be a compiled storyboard prompt, not a concept summary. Include a SCENES: section with exactly the requested number of concrete entries named SCENE_01, SCENE_02, etc. Every scene entry must include Visual/Action, Camera/Motion, Dialogue/VO (or [no dialogue]), Audio/SFX, and any visible text or reference usage for that scene. Do not provide only the source brief or generic layout instructions; malformed compiled storyboard prompts are blocked by quality audit."
|
|
302
317
|
},
|
|
303
318
|
"model": {
|
|
304
319
|
"type": "string",
|
|
@@ -306,9 +321,10 @@
|
|
|
306
321
|
"gpt-image-2",
|
|
307
322
|
"qwen-lightning",
|
|
308
323
|
"qwen",
|
|
309
|
-
"
|
|
324
|
+
"krea-identity-edit",
|
|
325
|
+
"dark-beast-krea2-identity-edit"
|
|
310
326
|
],
|
|
311
|
-
"description": "
|
|
327
|
+
"description": "The app auto-selects Fast→Qwen Lightning and HQ/Pro→full Qwen only for ordinary identity-neutral edits. REQUIRED IDENTITY DEFAULT: set \"krea-identity-edit\" whenever an edit of a referenced person or character must keep likeness or character identity while changing clothing, hair or makeup, pose or position, face/head/body, background, lighting, or visual style. Infer that semantic intent in any language; never route from keyword or regex matching. Also use it for a non-Pro single-character sheet. This default applies even when the user did not name Krea; an explicitly requested model always wins. Set \"dark-beast-krea2-identity-edit\" only when the user explicitly requests that model, its uncensored/community variant, or dark_beast_krea2_identity_edit_v1_2. Set \"gpt-image-2\" when the user explicitly names GPT/OpenAI/ChatGPT Image, or when precise typography, dense labels, or a professional multi-panel layout is the primary requirement; Pro character sheets may retain GPT Image 2. If GPT Image 2 is unavailable for detail-critical layout work, fall back to full \"qwen\", never \"qwen-lightning\". Krea identity edit models require at least one reference image, accept up to two context images, and work best at 512-2048px. Let the model tier and worker choose current steps, guidance, sampler, scheduler, grounding, and reference-boost defaults; do not send a negative prompt. When Krea is selected, override the generic prompt-length guidance with a concise 1-4 sentence delta instruction; name only the requested change and details that must remain fixed. Put the base scene/image first and an optional person/detail reference second. Z-image, Z-image Turbo, and base Krea 2 Turbo are generate_image img2img models, not edit_image selectors. If the user names another edit/image model, honor it. GPT Image 2 always processes input images at high fidelity; do not set input_fidelity."
|
|
312
328
|
},
|
|
313
329
|
"sourceImageIndex": {
|
|
314
330
|
"type": "number",
|
|
@@ -316,17 +332,17 @@
|
|
|
316
332
|
},
|
|
317
333
|
"numberOfVariations": {
|
|
318
334
|
"type": "number",
|
|
319
|
-
"description": "Number of variations (1-16). Pass the user's
|
|
335
|
+
"description": "Number of variations (1-16). Pass the user's exact requested count in one call when the outputs can share project settings. \"4 variations\" → numberOfVariations=4 in a single call. Use the exact requested count for reference-guided images that will feed a later video after the user picks one. Use separate calls only when the user explicitly wants independent projects, isolated approvals, or per-output settings that cannot share one project. For screenplay/storyboard batches, the prompt should contain one Dynamic Prompt branch with one full scene prompt per scene; do not set numberOfVariations=N with only scene 1's prompt. Use 1 unless the user explicitly asks for multiple. Default: 1.",
|
|
320
336
|
"minimum": 1,
|
|
321
337
|
"maximum": 16
|
|
322
338
|
},
|
|
323
339
|
"width": {
|
|
324
340
|
"type": "number",
|
|
325
|
-
"description": "Output image width in pixels. Defaults to the context image width. Supported
|
|
341
|
+
"description": "Output image width in pixels. Defaults to the context image width. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
326
342
|
},
|
|
327
343
|
"height": {
|
|
328
344
|
"type": "number",
|
|
329
|
-
"description": "Output image height in pixels. Defaults to the context image height. Supported
|
|
345
|
+
"description": "Output image height in pixels. Defaults to the context image height. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
|
|
330
346
|
},
|
|
331
347
|
"aspectRatio": {
|
|
332
348
|
"type": "string",
|
|
@@ -449,6 +465,38 @@
|
|
|
449
465
|
}
|
|
450
466
|
}
|
|
451
467
|
},
|
|
468
|
+
{
|
|
469
|
+
"type": "function",
|
|
470
|
+
"function": {
|
|
471
|
+
"name": "upscale_image",
|
|
472
|
+
"description": "Enlarge an existing image with NVIDIA RTX Video Super Resolution while preserving its content, identity, composition, and colors. This is deterministic reconstruction, not a generative edit: it takes no prompt and must not be used for restoration, sharpening requests that imply repainting, object changes, style changes, or creative enhancement. Use it when the user asks to upscale, enlarge, increase resolution, prepare for print, or produce a 2K/4K/6K/8K copy without changing the image.",
|
|
473
|
+
"parameters": {
|
|
474
|
+
"type": "object",
|
|
475
|
+
"properties": {
|
|
476
|
+
"sourceImageIndex": {
|
|
477
|
+
"type": "number",
|
|
478
|
+
"description": "Source image to upscale. Non-negative values select a prior generated result by 0-based index. Negative values select uploads: -1 is the first uploaded image, -2 the second, and so on. If omitted, use the latest generated image, falling back to the first upload."
|
|
479
|
+
},
|
|
480
|
+
"scale": {
|
|
481
|
+
"type": "number",
|
|
482
|
+
"enum": [
|
|
483
|
+
2,
|
|
484
|
+
3,
|
|
485
|
+
4
|
|
486
|
+
],
|
|
487
|
+
"description": "Edge scale multiplier. Use 2 by default. Ignored when targetLongestEdge is supplied. If that scale would leave either aligned output edge below 512px, the tool reports the minimum valid target instead of stretching the image."
|
|
488
|
+
},
|
|
489
|
+
"targetLongestEdge": {
|
|
490
|
+
"type": "number",
|
|
491
|
+
"minimum": 512,
|
|
492
|
+
"maximum": 8192,
|
|
493
|
+
"description": "Optional requested pixel length for the output longest edge, from 512 through 8192. Use 3840 for 4K UHD, 6144 for 6K, 7680 for 8K UHD, or 8192 for an 8K-class maximum. The other edge is calculated automatically so the source aspect ratio is preserved; both aligned output edges must be at least 512px."
|
|
494
|
+
}
|
|
495
|
+
},
|
|
496
|
+
"required": []
|
|
497
|
+
}
|
|
498
|
+
}
|
|
499
|
+
},
|
|
452
500
|
{
|
|
453
501
|
"type": "function",
|
|
454
502
|
"function": {
|
|
@@ -497,13 +545,13 @@
|
|
|
497
545
|
"type": "function",
|
|
498
546
|
"function": {
|
|
499
547
|
"name": "animate_photo",
|
|
500
|
-
"description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.3 or WAN 2.2. Do NOT use this tool for seedance2 or seedance2-
|
|
548
|
+
"description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.5 by default, LTX 2.3 as rollback, or WAN 2.2. Do NOT use this tool for seedance2, seedance2-mini, seedance2-fast, or seedance2-5. Seedance media references — including Seedance 2.5 first-and-last-frame requests — must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Avoid placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, prefer generating both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, then call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. If the user explicitly wants separately created frame assets, preserve that staged instruction while keeping indices correct. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Call once you have clear creative intent. N-VIDEOS PATTERN: Avoid sequential animate_photo calls for N outputs. For a single fixed source/end frame where only prompt text varies, use sourceImageIndex + numberOfVariations=N + one Dynamic Prompt branch in prompt so Sogni submits one project with multiple jobs. If the user explicitly asks for Dynamic Prompt or Dynamic Template syntax, prefer this one-project path whenever every output uses the same source/end frames and shared settings, even if they also ask to stitch the completed clips afterward. Use sourceImageIndices/prompts for multi-segment stitched non-Seedance video, different source/end assets, different audio windows, different durations/dimensions, isolated retry lifecycle, or other per-output parameters. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips on ltx23 (30s total = 2 clips × 15s) and 10-second clips on wan22 (60s total on wan22 = 6 clips × 10s; do NOT pick 4 clips × 15s on wan22 — the wan22 worker rejects clips longer than 10s). Multi-source flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image and only prompt text varies, use sourceImageIndex=-1, frameRole=\"both\" if requested, endImageIndex=-1 if requested, numberOfVariations=N, and one Dynamic Prompt branch in prompt. For a long scripted/dialogue/storyboard video from a single supplied/uploaded image where each segment needs isolated exact dialogue or per-segment wiring, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when source/end asset wiring or other per-output parameters differ, pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame.",
|
|
501
549
|
"parameters": {
|
|
502
550
|
"type": "object",
|
|
503
551
|
"properties": {
|
|
504
552
|
"prompt": {
|
|
505
553
|
"type": "string",
|
|
506
|
-
"description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.3 generates audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
|
|
554
|
+
"description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same source/end frames and generation parameters and only prompt text varies. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
|
|
507
555
|
},
|
|
508
556
|
"expandPrompt": {
|
|
509
557
|
"type": "boolean",
|
|
@@ -516,22 +564,33 @@
|
|
|
516
564
|
"videoModel": {
|
|
517
565
|
"type": "string",
|
|
518
566
|
"enum": [
|
|
567
|
+
"ltx25",
|
|
519
568
|
"ltx23",
|
|
520
|
-
"wan22"
|
|
569
|
+
"wan22",
|
|
570
|
+
"happyhorse-1.1-i2v",
|
|
571
|
+
"happyhorse-1.1-r2v",
|
|
572
|
+
"minimax-h3-i2v",
|
|
573
|
+
"minimax-h3-i2v-turbo",
|
|
574
|
+
"minimax-h3-flf2v",
|
|
575
|
+
"minimax-h3-flf2v-turbo"
|
|
521
576
|
],
|
|
522
|
-
"description": "Which video model to use. \"
|
|
577
|
+
"description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, seedance2-fast, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
|
|
578
|
+
},
|
|
579
|
+
"generateAudio": {
|
|
580
|
+
"type": "boolean",
|
|
581
|
+
"description": "Whether to include generated/native audio for audio-capable models. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. When false, the returned video has no audio track. Ignored by audio-less WAN."
|
|
523
582
|
},
|
|
524
583
|
"negativePrompt": {
|
|
525
584
|
"type": "string",
|
|
526
|
-
"description": "
|
|
585
|
+
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. All standard LTX 2.5 image and first/last-frame workflows accept this separate negative prompt. Use this field only when the user explicitly asks to set one. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt."
|
|
527
586
|
},
|
|
528
587
|
"duration": {
|
|
529
588
|
"type": "number",
|
|
530
|
-
"description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\").
|
|
589
|
+
"description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Per-model maximum: ltx25 and ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip."
|
|
531
590
|
},
|
|
532
591
|
"targetResolution": {
|
|
533
592
|
"type": "number",
|
|
534
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
593
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
535
594
|
},
|
|
536
595
|
"sourceImageIndex": {
|
|
537
596
|
"type": "number",
|
|
@@ -544,7 +603,7 @@
|
|
|
544
603
|
},
|
|
545
604
|
"minItems": 1,
|
|
546
605
|
"maxItems": 16,
|
|
547
|
-
"description": "Array of source frame indices — one video is generated per entry as its own SDK project, all running in PARALLEL. Use 0-based non-negative result indices for generated images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Repeating -1 is allowed
|
|
606
|
+
"description": "Array of source frame indices — one video is generated per entry as its own SDK project, all running in PARALLEL. Use this when outcomes need different source images, different end frames, isolated retry lifecycle, or other per-clip asset wiring/parameters. If every outcome uses the same source/end frames and only prompt text differs, prefer sourceImageIndex with numberOfVariations=N and one Dynamic Prompt branch in `prompt` so Sogni creates one project with multiple jobs. Use 0-based non-negative result indices for generated images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Repeating -1 is allowed for true multi-project workflows that intentionally reuse the same uploaded image while varying per-clip assets or parameters. By default all projects share the `prompt`/`voice`/`duration`, but you can pass `prompts` (array) to give each clip its own dialogue/motion when multi-project fan-out is required. Avoid sequential animate_photo calls for N outputs. Do NOT combine with `numberOfVariations` or `sourceImageIndex`. Use frameRole=\"end\" with sourceImageIndices only when the user explicitly says the repeated uploaded/generated image is the last/end frame for each clip and no first/start frame should be supplied; in that case omit endImageIndex/endImageIndices because each sourceImageIndices entry is the end frame. You MAY combine with frameRole=\"both\" when clips need start and end frames. For adjacent transition chains across generated images, use sourceImageIndices=[start..end-1] and endImageIndices=[start+1..end] so N images produce N-1 transition clips. If the uploaded/original image starts the chain and generated results are the remaining frames, use sourceImageIndices=[-1,start..end-1] and endImageIndices=[start..end]. If the user supplies multiple uploaded images as the actual keyframe sequence, use adjacent negative uploaded indices, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4], endImageIndices=[-2,-3,-4,-5], frameRole=\"both\", prompts length 4, then stitch_video. If the user specifies transition motion, camera behavior, actions, dialogue, or audio, copy those instructions into every corresponding per-clip prompt; only invent a generic smooth transition when the user does not specify one. If the user asks for a seamless loop or final transition from the last image back to the first, close the chain by including the last image as a source and the first image as the final end frame, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4,-5], endImageIndices=[-2,-3,-4,-5,-1]. For generated scene keyframes that should each loop to themselves, omit endImageIndex/endImageIndices so each source image is also its own end frame. Set endImageIndex=-1 only when every sourceImageIndices entry is also -1 and every segment reuses the first uploaded image. Range: 1–16 indices. For generated image batches, values MUST be read from the latest edit_image/generate_image tool result's `startIndex` field. If startIndex=3 and 4 images were generated in that batch, pass `[3,4,5,6]` (NOT `[0,1,2,3]`). Do NOT assume generated indices start at 0 — they don't if there are prior results in the conversation."
|
|
548
607
|
},
|
|
549
608
|
"prompts": {
|
|
550
609
|
"type": "array",
|
|
@@ -553,11 +612,11 @@
|
|
|
553
612
|
},
|
|
554
613
|
"minItems": 1,
|
|
555
614
|
"maxItems": 16,
|
|
556
|
-
"description": "Per-clip prompts for fan-out — use when
|
|
615
|
+
"description": "Per-clip prompts for multi-project fan-out — use when each output needs different source/end assets, isolated retry lifecycle, or other per-clip wiring/parameters. If all outputs share the same source/end frames and only prompt text differs, put the full per-output prompts in ONE Dynamic Prompt branch in `prompt` and set numberOfVariations=N instead. When this field is required, it MUST be paired with `sourceImageIndices` and have the SAME length. Each entry is the full prompt for the corresponding source image. If a clip has speech, include exact spoken words in double quotes with stable speaker tags; do NOT write placeholders like \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". If you just wrote a script/table/storyboard, copy that clip's exact dialogue into this prompt. When named speakers appear in a multi-person reference image or generated keyframe, start each entry with one compact cast map that binds names to visible anchors before dialogue, e.g. Cast map: SPEAKER_A is the left person holding the prop; SPEAKER_B is the center person with the tablet; SPEAKER_C is the right person near the table. Then move directly into action/dialogue; do not describe those same people again as generic man/boy/girl/woman/character subjects. This prevents speaker tags from being assigned to the wrong visible character. When set, the top-level `prompt` parameter is ignored (still required by the schema — just pass any descriptive string, e.g. a brief summary of the batch). Example: 4 source images of a couple, \"make each video have a different joke\" → sourceImageIndices=[0,1,2,3], prompts=[\"Cast map: She is the left woman in the blue dress; He is the right man in the gray jacket. She says: \\\"Why did the scarecrow win an award?\\\" He grins.\", \"Cast map: He is the right man in the gray jacket; She is the left woman in the blue dress. He says: \\\"Because he was outstanding in his field!\\\" She laughs.\", \"...\", \"...\"]. Omit this whenever the same source/end assets and parameters can be represented as one Dynamic Prompt batch."
|
|
557
616
|
},
|
|
558
617
|
"numberOfVariations": {
|
|
559
618
|
"type": "number",
|
|
560
|
-
"description": "Number of variations (1-16). Use 1 unless user explicitly requests multiple separate video outputs.",
|
|
619
|
+
"description": "Number of variations (1-16). Use this with one Dynamic Prompt branch when the user explicitly requests multiple prompt-only takes from the same source/end frames. This creates one Sogni project with multiple jobs. Use 1 unless the user explicitly requests multiple separate video outputs; use sourceImageIndices/prompts instead only when assets or parameters differ per output.",
|
|
561
620
|
"minimum": 1,
|
|
562
621
|
"maximum": 16
|
|
563
622
|
},
|
|
@@ -572,7 +631,7 @@
|
|
|
572
631
|
"end",
|
|
573
632
|
"both"
|
|
574
633
|
],
|
|
575
|
-
"description": "How to use the source image(s) for non-Seedance video generation. \"start\" (default): image is the first frame — video animates forward from it. \"end\": image is the last frame — video leads up to it. \"both\": two images provided — interpolates between start and end frames. For single clips using \"both\", set sourceImageIndex to the start frame and endImageIndex to the end frame. For sourceImageIndices fan-out using repeated -1, use frameRole=\"both\" and set endImageIndex=-1 when every segment must use the uploaded image as
|
|
634
|
+
"description": "How to use the source image(s) for non-Seedance video generation. \"start\" (default): image is the first frame — video animates forward from it. \"end\": image is the last frame — video leads up to it. \"both\": two images provided — interpolates between start and end frames. For single clips using \"both\", set sourceImageIndex to the start frame and endImageIndex to the end frame. For sourceImageIndices fan-out using repeated -1, use frameRole=\"end\" only when every clip should use that image as the last/end frame and no first/start frame should be supplied. Use frameRole=\"both\" and set endImageIndex=-1 when every segment must use the uploaded image as both its first and last frame. For adjacent generated-image transitions, use frameRole=\"both\" with matching sourceImageIndices and endImageIndices arrays. Omit endImageIndex/endImageIndices only when each source image should also be its own end frame. The handler inspects different start/end frames and generates a detailed transition prompt automatically."
|
|
576
635
|
},
|
|
577
636
|
"endImageIndex": {
|
|
578
637
|
"type": "number",
|
|
@@ -589,7 +648,7 @@
|
|
|
589
648
|
},
|
|
590
649
|
"voicePersonaName": {
|
|
591
650
|
"type": "string",
|
|
592
|
-
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23."
|
|
651
|
+
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
|
|
593
652
|
}
|
|
594
653
|
},
|
|
595
654
|
"required": [
|
|
@@ -633,13 +692,13 @@
|
|
|
633
692
|
"type": "function",
|
|
634
693
|
"function": {
|
|
635
694
|
"name": "video_to_video",
|
|
636
|
-
"description": "Transform an existing video using
|
|
695
|
+
"description": "Transform an existing video using WAN 2.2 Animate, LTX 2.5 V2V controls by default, LTX 2.3 as rollback, or Seedance V2V when explicitly requested. LTX 2.5 distilled supports canny/pose/depth/detailer/inpaint/outpaint; Dev + Speed LoRA supports canny/pose/depth/detailer. Requires an uploaded video.",
|
|
637
696
|
"parameters": {
|
|
638
697
|
"type": "object",
|
|
639
698
|
"properties": {
|
|
640
699
|
"prompt": {
|
|
641
700
|
"type": "string",
|
|
642
|
-
"description": "Describe the TARGET appearance
|
|
701
|
+
"description": "Describe the TARGET appearance, motion, dialogue, audio, and style in positive present-tense language. For LTX 2.5 (default) or LTX 2.3 rollback canny/depth/pose modes, the source preserves the selected structure or motion, so emphasize style, atmosphere, lighting, texture, color, scale, and pacing. Canny preserves edges; pose preserves skeletal motion; depth preserves 3D layout; detailer should describe the original content with quality qualifiers only. Distilled LTX 2.5 also supports inpaint and outpaint; Dev + Speed LoRA does not. For inpaint, describe only the regenerated region. For outpaint, describe the newly revealed area consistently with the source. For Seedance V2V, use natural prose and describe the target transformation holistically."
|
|
643
702
|
},
|
|
644
703
|
"expandPrompt": {
|
|
645
704
|
"type": "boolean",
|
|
@@ -660,29 +719,32 @@
|
|
|
660
719
|
"detailer",
|
|
661
720
|
"seedance-v2v"
|
|
662
721
|
],
|
|
663
|
-
"description": "How the source video and (optional) reference image interact. Pick by user intent:\n• \"animate-move\" (DEFAULT) — WAN 2.2 Animate Move. Applies camera movement and motion from the source video to the reference image, bringing a still photo to life. Requires sourceImageIndex.\n• \"animate-replace\" — WAN 2.2 Animate Replace. Replaces the subject in the source video with the person/character from the reference image, keeping the video's background and motion. Requires sourceImageIndex.\n• \"canny\" — LTX-2.3 edge-detection control. Best for restyling while preserving exact composition and silhouettes (e.g. \"make this footage look like anime / oil painting / watercolor\"). Use for subjects with crisp edges — people, objects, graphics. Video-only; no reference image needed.\n• \"pose\" — LTX-2.3 skeletal tracking. Best for replacing a person while keeping their motion (e.g. \"turn this dancer into a robot\"). Image optional — if provided, controls appearance; otherwise the prompt drives appearance. Requires person-centric motion.\n• \"depth\" — LTX-2.3 depth-map control. Best for restyling scenes with perspective, camera movement, or volumetric content (landscapes, interiors, camera pans). Preserves 3D spatial layout rather than 2D edges; more forgiving than canny when edges are noisy. Video-only.\n• \"detailer\" — LTX-2.3 quality enhancement. Sharpens detail and texture WITHOUT restyling. The prompt must DESCRIBE THE ORIGINAL scene with quality qualifiers (sharp, clean, high-resolution) — never request content changes, new textures, or a new look. Pick this when the user asks to \"improve quality\", \"enhance\", \"upscale\", or \"sharpen\" without a creative transformation.\n• \"seedance-v2v\" — BytePlus Dreamina Seedance
|
|
722
|
+
"description": "How the source video and (optional) reference image interact. Pick by user intent:\n• \"animate-move\" (DEFAULT) — WAN 2.2 Animate Move. Applies camera movement and motion from the source video to the reference image, bringing a still photo to life. Requires sourceImageIndex.\n• \"animate-replace\" — WAN 2.2 Animate Replace. Replaces the subject in the source video with the person/character from the reference image, keeping the video's background and motion. Requires sourceImageIndex.\n• \"canny\" — LTX-2.3 edge-detection control. Best for restyling while preserving exact composition and silhouettes (e.g. \"make this footage look like anime / oil painting / watercolor\"). Use for subjects with crisp edges — people, objects, graphics. Video-only; no reference image needed.\n• \"pose\" — LTX-2.3 skeletal tracking. Best for replacing a person while keeping their motion (e.g. \"turn this dancer into a robot\"). Image optional — if provided, controls appearance; otherwise the prompt drives appearance. Requires person-centric motion.\n• \"depth\" — LTX-2.3 depth-map control. Best for restyling scenes with perspective, camera movement, or volumetric content (landscapes, interiors, camera pans). Preserves 3D spatial layout rather than 2D edges; more forgiving than canny when edges are noisy. Video-only.\n• \"detailer\" — LTX-2.3 quality enhancement. Sharpens detail and texture WITHOUT restyling. The prompt must DESCRIBE THE ORIGINAL scene with quality qualifiers (sharp, clean, high-resolution) — never request content changes, new textures, or a new look. Pick this when the user asks to \"improve quality\", \"enhance\", \"upscale\", or \"sharpen\" without a creative transformation.\n• \"seedance-v2v\" — BytePlus Dreamina Seedance video-to-video. Use only when the user explicitly asks for Seedance on the uploaded source video, such as Seedance Fast upscale, enhance, remaster, restyle, or transform. High-fidelity quality, native audio, time-coded scene control. Seedance V2V reads @Video1 holistically. Use it for restyling, motion transfer, extension, subject replacement, or scene transformation, and assign @Video1 a clear role such as source clip, camera movement, action timing, edit rhythm, or continuation anchor. Distinct from canny/depth/pose which use control-net constraints — Seedance treats the reference video holistically.\nCanny vs depth: canny preserves silhouettes and fine outlines — pick it for subject-led scenes and graphic restyles. Depth preserves 3D structure — pick it for scenes where the camera moves or spatial layout matters more than edge fidelity. Default: \"animate-move\"."
|
|
664
723
|
},
|
|
665
724
|
"negativePrompt": {
|
|
666
725
|
"type": "string",
|
|
667
|
-
"description": "Non-Seedance only. Optional negative prompt
|
|
726
|
+
"description": "Non-Seedance only. Optional negative prompt supported by every LTX 2.5 and LTX 2.3 video-to-video control/edit template, plus WAN. Do not set when controlMode is seedance-v2v or videoModel is seedance2/seedance2-mini/seedance2-fast/seedance2-5; rewrite user-provided Seedance avoid/ban/no-X requests as positive prompt instructions."
|
|
668
727
|
},
|
|
669
728
|
"videoModel": {
|
|
670
729
|
"type": "string",
|
|
671
730
|
"enum": [
|
|
731
|
+
"ltx25-v2v",
|
|
672
732
|
"ltx23-v2v",
|
|
673
733
|
"wan22-animate",
|
|
674
734
|
"seedance2",
|
|
675
|
-
"seedance2-
|
|
735
|
+
"seedance2-mini",
|
|
736
|
+
"seedance2-fast",
|
|
737
|
+
"seedance2-5"
|
|
676
738
|
],
|
|
677
|
-
"description": "Model selector for this video-to-video request. Usually omit; controlMode chooses the non-Seedance model. For controlMode=\"seedance-v2v\", use \"seedance2-
|
|
739
|
+
"description": "Model selector for this video-to-video request. Usually omit; controlMode chooses the non-Seedance model. \"ltx25-v2v\" is the default LTX 2.5 control path; \"ltx23-v2v\" remains the rollback selector. For controlMode=\"seedance-v2v\", Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance V2V unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro or from 480p/720p resolution requests alone. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
|
|
678
740
|
},
|
|
679
741
|
"generateAudio": {
|
|
680
742
|
"type": "boolean",
|
|
681
|
-
"description": "
|
|
743
|
+
"description": "Whether the final video should include generated or retained audio. Omit to include audio by default; set false when the user asks for silent output or no audio. When false, the returned video has no audio track."
|
|
682
744
|
},
|
|
683
745
|
"targetResolution": {
|
|
684
746
|
"type": "number",
|
|
685
|
-
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
747
|
+
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V Mini, Fast, and Seedance 2.5 support 480p and 720p only, so never set 1080p or 4K for \"seedance2-5\". Preserve the source video shape instead of forcing landscape pixels."
|
|
686
748
|
},
|
|
687
749
|
"sourceImageIndex": {
|
|
688
750
|
"type": "number",
|
|
@@ -690,9 +752,9 @@
|
|
|
690
752
|
},
|
|
691
753
|
"duration": {
|
|
692
754
|
"type": "number",
|
|
693
|
-
"description": "Output video duration in seconds. Range: 2-20 for WAN/LTX modes and 4-
|
|
755
|
+
"description": "Output video duration in seconds. Range: 2-20 for WAN/LTX modes, 4-15 for controlMode=\"seedance-v2v\" on \"seedance2\"/\"seedance2-mini\"/\"seedance2-fast\", and 4-30 for controlMode=\"seedance-v2v\" on \"seedance2-5\". If omitted, the tool matches the uploaded source video duration when available (capped to the selected model range); otherwise it falls back to 10s for WAN Animate Move/Replace and 5s for LTX-2.3/Seedance modes. For long stitched/bulk WAN Animate Move/Replace work with no explicit per-clip length, prefer about 10s clips rather than 5s chunks. Only pass this when the user explicitly requests a different length.",
|
|
694
756
|
"minimum": 2,
|
|
695
|
-
"maximum":
|
|
757
|
+
"maximum": 30
|
|
696
758
|
},
|
|
697
759
|
"numberOfVariations": {
|
|
698
760
|
"type": "number",
|
|
@@ -711,7 +773,7 @@
|
|
|
711
773
|
"type": "function",
|
|
712
774
|
"function": {
|
|
713
775
|
"name": "stitch_video",
|
|
714
|
-
"description": "
|
|
776
|
+
"description": "Concatenate whole videos end-to-end into one continuous video. This tool joins each source clip in full, in the order you pass — it does NOT interleave time slices, insert one clip inside another, or replace part of a video. WHEN TO USE: the user wants clips played one after another (whole clip A, then whole clip B), including adding a generated bumper / intro / outro / tag / sting before or after another video. Plain language: \"stitch these together\", \"stitch A and B\", \"combine these clips\", \"join these into one video\", \"play the bumper before this clip\". WHEN NOT TO USE — prefer replace_video_segment instead: any request to put one clip inside another, replace a window inside a video, alternate / interleave / splice short slices of multiple videos, insert clip X \"into the middle of\" clip Y, or swap out part of an existing video while keeping the rest. The word \"stitch\" in the user request does not by itself decide this tool — read what they actually want. If the user asks to \"stitch X into the middle of Y\" or \"stitch X into Y starting at 5s\", that is splice-into-middle and belongs to replace_video_segment. SOURCES: previously generated clips (non-negative indices into the session video-result array, populated by animate_photo, generate_video, sound_to_video, video_to_video, dance_montage — use videoStartIndex from their results to find the indices) and/or uploaded videos (negative indices: -1 = first uploaded video, -2 = second, etc.). Mix and match in any playback order — for example, pass [0, -1] to play the first generated clip followed by the first uploaded video (a generated bumper followed by the user's existing footage). When the user asks to stitch \"these\" or all uploaded videos and does not name a different playback order, use the current upload/UI order exactly: [-1, -2, ...]. If the user explicitly asks for a different order, honor that requested order. Requires at least 2 source videos in total. Never ask the user to re-upload videos that were already generated or that are already attached to the session. When the user generated music with generate_music in this same session and wants it on the stitch (or asked for a music video / soundtrack), pass a non-negative audioIndex to attach that generated track. When the user uploaded an audio file and wants it overlaid on the stitched video (e.g. \"stitch the audio after\", \"overlay the audio\", \"audio on top of the video\"), pass a negative audioIndex (-1 = first uploaded audio, -2 = second, etc.). In both cases the source clips' own audio is replaced by the chosen track. When the user asks for a fade, dissolve, wipe, or slide between clips, pass `transition`; omit `transition` for a hard cut (the default).",
|
|
715
777
|
"parameters": {
|
|
716
778
|
"type": "object",
|
|
717
779
|
"properties": {
|
|
@@ -883,25 +945,29 @@
|
|
|
883
945
|
"type": "function",
|
|
884
946
|
"function": {
|
|
885
947
|
"name": "sound_to_video",
|
|
886
|
-
"description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx23-ia2v; for lip-sync with a face image, use wan-s2v; if no image, use ltx23-a2v. If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.3 generates audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video
|
|
948
|
+
"description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx25-ia2v by default (ltx23-ia2v is rollback); for lip-sync with a face image, use wan-s2v; if no image, use ltx25-a2v by default (ltx23-a2v is rollback). If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video with videoModel=\"ltx23\" because LTX 2.5 has no compatible ID-LoRA. LONG AUDIO ON SEEDANCE: Seedance 2.0, Mini, and Fast cap each clip at 15s; Seedance 2.5 caps each clip at 30s, so prefer \"seedance2-5\" for 16-30s audio instead of splitting. When uploaded audio exceeds the selected Seedance model's per-clip cap, do NOT clamp and drop the rest — split the run into multiple sound_to_video calls in the same turn using 15s segments for seedance2/seedance2-mini/seedance2-fast or 30s segments for seedance2-5, then finish with a single stitch_video call referencing the resulting clip indices in order with audioIndex pointing at the same uploaded audio so the stitched output carries the full original soundtrack. LTX/WAN models accept up to 20s per clip, so single-call is fine for them.",
|
|
887
949
|
"parameters": {
|
|
888
950
|
"type": "object",
|
|
889
951
|
"properties": {
|
|
890
952
|
"prompt": {
|
|
891
953
|
"type": "string",
|
|
892
|
-
"description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.3 image+audio mode, do not re-describe static details already visible in the reference image; focus on motion, action, camera, and how the image responds to the audio.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main visual beat + 1 simple camera move. Around 10s: 2-3 clear beats + 1 camera move. >10s: up to 4 beats in clear sequence. Let the audio define timing, but avoid stacking subject, camera, and environment motion in short clips.\n\nBLOCKING: Direct layout when it affects the shot: left/right placement, foreground/background, facing direction, and relative distance between subjects.\n\nLIP-SYNC: Shot framing, speaker's appearance and setting, physical performance synced to audio — gestures, expressions, jaw movement between phrases. Include acting beats.\n\nMUSIC VISUALIZATION: Visual style, environment, and how elements react to rhythm and energy.\n\nAUDIO-REACTIVE: Motion and visual changes that correspond to sounds in the track.\n\nLTX VOCABULARY: camera (tracking, dolly, pan, tilt, handheld, static frame), lighting/atmosphere (golden hour, neon glow, dramatic shadows, fog, rain, smoke, reflections), scale/pacing (expansive, epic, intimate, claustrophobic, slow motion, time-lapse, lingering shot, continuous shot), style/genre (film noir, painterly, cyberpunk, stop-motion, claymation, 2D/3D animation, hand-drawn, fantasy, thriller, experimental film).\n\nAVOID: Vague prompts, too many competing visual elements, abstract descriptions without visible behavior, rigid numeric constraints, readable text or logos. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the visual interpretation while keeping audio sync intent consistent. Example: \"{abstract neon visualization|nature scene with swaying trees|urban street with rain} synced to the beat\"."
|
|
954
|
+
"description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.5 or LTX 2.3 image+audio mode, do not re-describe static details already visible in the reference image; focus on motion, action, camera, and how the image responds to the audio.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main visual beat + 1 simple camera move. Around 10s: 2-3 clear beats + 1 camera move. >10s: up to 4 beats in clear sequence. Let the audio define timing, but avoid stacking subject, camera, and environment motion in short clips.\n\nBLOCKING: Direct layout when it affects the shot: left/right placement, foreground/background, facing direction, and relative distance between subjects.\n\nLIP-SYNC: Shot framing, speaker's appearance and setting, physical performance synced to audio — gestures, expressions, jaw movement between phrases. Include acting beats.\n\nMUSIC VISUALIZATION: Visual style, environment, and how elements react to rhythm and energy.\n\nAUDIO-REACTIVE: Motion and visual changes that correspond to sounds in the track.\n\nLTX VOCABULARY: camera (tracking, dolly, pan, tilt, handheld, static frame), lighting/atmosphere (golden hour, neon glow, dramatic shadows, fog, rain, smoke, reflections), scale/pacing (expansive, epic, intimate, claustrophobic, slow motion, time-lapse, lingering shot, continuous shot), style/genre (film noir, painterly, cyberpunk, stop-motion, claymation, 2D/3D animation, hand-drawn, fantasy, thriller, experimental film).\n\nAVOID: Vague prompts, too many competing visual elements, abstract descriptions without visible behavior, rigid numeric constraints, readable text or logos. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the visual interpretation while keeping audio sync intent consistent. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same audio source/window, image source, model, duration, dimensions, and parameters and only prompt text varies. Example: \"{abstract neon visualization|nature scene with swaying trees|urban street with rain} synced to the beat\"."
|
|
893
955
|
},
|
|
894
956
|
"expandPrompt": {
|
|
895
957
|
"type": "boolean",
|
|
896
958
|
"description": "Seedance only. Whether to run the shared Seedance prompt shaper before dispatch. Defaults to true; set false only when the user explicitly asks to submit the compact prompt directly or not modify the prompt."
|
|
897
959
|
},
|
|
960
|
+
"negativePrompt": {
|
|
961
|
+
"type": "string",
|
|
962
|
+
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. The LTX A2V and IA2V workflows accept this separate negative prompt. Use it only when the user explicitly asks to set one. Do not set for Seedance."
|
|
963
|
+
},
|
|
898
964
|
"audioSourceIndex": {
|
|
899
965
|
"type": "number",
|
|
900
966
|
"description": "Index of the uploaded audio file to use (0-based, from uploaded files list). If only one audio file is uploaded, use 0. If no audio was uploaded but generate_music was used earlier, omit this — the tool will automatically find the generated audio."
|
|
901
967
|
},
|
|
902
968
|
"sourceImageIndex": {
|
|
903
969
|
"type": "number",
|
|
904
|
-
"description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.3 A2V), this is optional — omit it to generate video purely from text + audio."
|
|
970
|
+
"description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.5 or LTX 2.3 A2V), this is optional — omit it to generate video purely from text + audio."
|
|
905
971
|
},
|
|
906
972
|
"audioStart": {
|
|
907
973
|
"type": "number",
|
|
@@ -910,34 +976,38 @@
|
|
|
910
976
|
},
|
|
911
977
|
"duration": {
|
|
912
978
|
"type": "number",
|
|
913
|
-
"description": "Video duration in seconds. Default: 5. Range: 2-20. For music videos, use the MAXIMUM duration (20) since the audio is always longer than the video limit. Use when the user explicitly requests a specific length.",
|
|
979
|
+
"description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.5, LTX 2.3, and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30. For music videos, use the MAXIMUM duration the selected model allows (20 for LTX/WAN, 30 for \"seedance2-5\") since the audio is always longer than the video limit. Use when the user explicitly requests a specific length.",
|
|
914
980
|
"minimum": 2,
|
|
915
|
-
"maximum":
|
|
981
|
+
"maximum": 30
|
|
916
982
|
},
|
|
917
983
|
"videoModel": {
|
|
918
984
|
"type": "string",
|
|
919
985
|
"enum": [
|
|
920
986
|
"wan-s2v",
|
|
921
987
|
"seedance2",
|
|
988
|
+
"seedance2-mini",
|
|
922
989
|
"seedance2-fast",
|
|
990
|
+
"seedance2-5",
|
|
991
|
+
"ltx25-ia2v",
|
|
992
|
+
"ltx25-a2v",
|
|
923
993
|
"ltx23-ia2v",
|
|
924
994
|
"ltx23-a2v"
|
|
925
995
|
],
|
|
926
|
-
"description": "Video model. \"
|
|
996
|
+
"description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast/HQ use official distilled INT8 and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. \"seedance2-fast\": legacy Seedance 2.0 Fast. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance unless the user explicitly says Seedance Fast, pick \"seedance2-fast\" when the user says Seedance Fast / seedance-fast, and pick \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
|
|
927
997
|
},
|
|
928
998
|
"generateAudio": {
|
|
929
999
|
"type": "boolean",
|
|
930
|
-
"description": "
|
|
1000
|
+
"description": "Whether the final video should include audio. Omit to include audio by default; set false when the user asks for silent output or no audio. When false, the returned video has no audio track; the reference audio is still required and still drives generation."
|
|
931
1001
|
},
|
|
932
1002
|
"numberOfVariations": {
|
|
933
1003
|
"type": "number",
|
|
934
|
-
"description": "Number of video variations to generate (1-16). Default: 1.",
|
|
1004
|
+
"description": "Number of video variations to generate (1-16). Use with one Dynamic Prompt branch when all variations share the same audio source/window, image source, model, duration, dimensions, and parameters and only prompt text varies. This creates one Sogni project with multiple jobs. Default: 1.",
|
|
935
1005
|
"minimum": 1,
|
|
936
1006
|
"maximum": 16
|
|
937
1007
|
},
|
|
938
1008
|
"targetResolution": {
|
|
939
1009
|
"type": "number",
|
|
940
|
-
"description": "Short-side video resolution target in pixels. Use
|
|
1010
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
941
1011
|
},
|
|
942
1012
|
"aspectRatio": {
|
|
943
1013
|
"type": "string",
|
|
@@ -954,7 +1024,7 @@
|
|
|
954
1024
|
"type": "function",
|
|
955
1025
|
"function": {
|
|
956
1026
|
"name": "extend_video",
|
|
957
|
-
"description": "Extend a video by adding new time to the end. Works on BOTH videos previously rendered in this session AND user-uploaded videos — set videoIndex to a negative number (e.g. -1) to target an uploaded video when no prior render exists. The base video is auto-selected from the most recent video in this session unless videoIndex is set. For LTX-2.3 base clips, the tool extracts the last frame and renders an image-to-video continuation. For Seedance base clips, the tool extracts a trailing reference segment and renders a video-to-video continuation. Returns both the standalone new segment and a spliced composite (base + new segment). Use when the user asks to \"make it longer\", \"extend the video\", \"add another N seconds\", \"continue the scene\", \"add an outro/bumper to the end\", etc. Prefer this over generate_image+animate_photo+stitch_video for \"add a bumper/outro to this video\" — extend_video preserves the original base bytes, audio, and timing instead of re-encoding them. Do not use this tool to render fresh videos from scratch — call generate_video or animate_photo for that. Output durations follow each model's native limits (LTX 2-20s, Seedance 4-15s) for the new segment alone.",
|
|
1027
|
+
"description": "Extend a video by adding new time to the end. Works on BOTH videos previously rendered in this session AND user-uploaded videos — set videoIndex to a negative number (e.g. -1) to target an uploaded video when no prior render exists. The base video is auto-selected from the most recent video in this session unless videoIndex is set. For LTX-2.3 base clips, the tool extracts the last frame and renders an image-to-video continuation. For Seedance base clips, the tool extracts a trailing reference segment and renders a video-to-video continuation. Returns both the standalone new segment and a spliced composite (base + new segment). Use when the user asks to \"make it longer\", \"extend the video\", \"add another N seconds\", \"continue the scene\", \"add an outro/bumper to the end\", etc. Prefer this over generate_image+animate_photo+stitch_video for \"add a bumper/outro to this video\" — extend_video preserves the original base bytes, audio, and timing instead of re-encoding them. Do not use this tool to render fresh videos from scratch — call generate_video or animate_photo for that. Output durations follow each model's native limits (LTX 2-20s, Seedance 2.0/Mini/Fast 4-15s, Seedance 2.5 4-30s) for the new segment alone.",
|
|
958
1028
|
"parameters": {
|
|
959
1029
|
"type": "object",
|
|
960
1030
|
"properties": {
|
|
@@ -964,9 +1034,9 @@
|
|
|
964
1034
|
},
|
|
965
1035
|
"duration": {
|
|
966
1036
|
"type": "number",
|
|
967
|
-
"description": "Length in seconds of the new appended segment (NOT total final length). LTX 2-20, Seedance 4-15. Default: 5.",
|
|
1037
|
+
"description": "Length in seconds of the new appended segment (NOT total final length). LTX 2-20, Seedance 2.0/Mini/Fast 4-15, Seedance 2.5 4-30. Default: 5.",
|
|
968
1038
|
"minimum": 2,
|
|
969
|
-
"maximum":
|
|
1039
|
+
"maximum": 30
|
|
970
1040
|
},
|
|
971
1041
|
"videoIndex": {
|
|
972
1042
|
"type": "number",
|
|
@@ -976,11 +1046,14 @@
|
|
|
976
1046
|
"type": "string",
|
|
977
1047
|
"enum": [
|
|
978
1048
|
"auto",
|
|
1049
|
+
"ltx25",
|
|
979
1050
|
"ltx23",
|
|
980
1051
|
"seedance2",
|
|
981
|
-
"seedance2-
|
|
1052
|
+
"seedance2-mini",
|
|
1053
|
+
"seedance2-fast",
|
|
1054
|
+
"seedance2-5"
|
|
982
1055
|
],
|
|
983
|
-
"description": "Which model to use for the new segment. Default: \"auto\" —
|
|
1056
|
+
"description": "Which model to use for the new segment. Default: \"auto\" — preserve Seedance for a Seedance base and otherwise use LTX 2.5. Use ltx23 only for explicit rollback. \"seedance2-5\" supports 480p/720p and 4-30s of new footage at 24 fps."
|
|
984
1057
|
},
|
|
985
1058
|
"keepOriginalAudio": {
|
|
986
1059
|
"type": "boolean",
|
|
@@ -997,7 +1070,7 @@
|
|
|
997
1070
|
"type": "function",
|
|
998
1071
|
"function": {
|
|
999
1072
|
"name": "replace_video_segment",
|
|
1000
|
-
"description": "
|
|
1073
|
+
"description": "Modify a portion of an existing video while keeping the rest intact — either by regenerating that slice fresh or by splicing in another existing clip. Operates on a [startSeconds, endSeconds] window inside a base video; everything outside the window stays exactly as it was. Works on BOTH videos previously rendered in this session AND user-uploaded videos (set videoIndex to a negative number to target an uploaded video when no prior render exists). WHEN TO USE: any request to change part of one video while keeping the rest, to put another clip inside another video at a specific position, or to interleave time slices of multiple videos. Plain language: \"regenerate from 5s to 10s\", \"redo the last 3 seconds\", \"swap out the middle\", \"replace the bumper at the end\", \"swap the end card\", \"change the outro / intro / ending / last clip\", \"replace 2s-4s with a stronger expression\", \"splice video 2 into video 1\", \"stitch video 2 into the middle of video 1\", \"insert the second clip at 5s\", \"alternate 1 second from each video\". The word \"stitch\" in the user request does not by itself mean stitch_video — when the user clearly wants insertion or in-place replacement, this tool is the right one. WHEN NOT TO USE — prefer stitch_video instead: the user wants to concatenate whole clips end-to-end without modifying their interiors (\"stitch these together\", \"play A then B\", \"add a bumper before / after\"). SPLICING EXISTING CLIPS: pass replacementVideoIndex when the replacement already exists as an uploaded or generated video — do not call generate_video / animate_photo / video_to_video in that case. Set endSeconds=startSeconds when the user asks for an insertion that should not remove time from the base video. TIME-SLICED INTERLEAVING (\"alternate 1 second from each video\"): pass replacementStartSeconds and replacementEndSeconds to cut the next source slice out of the replacement video before splicing it into the base. Repeat this call for each alternating window. By default use replacement windows (endSeconds = startSeconds + sliceDuration); use insertion windows (endSeconds = startSeconds) only when the user explicitly asks to lengthen the output by inserting extra slices. replacementStartSeconds and replacementEndSeconds must be concrete non-negative seconds; never use -1 as an end-of-source sentinel. PREFER this over re-running generate_video / animate_photo on the original prompt when the user only wants part of the video changed — re-rendering wastes credits, loses the unchanged sections, and breaks the original timing. If the user does not specify the exact start/end seconds (e.g. \"replace the bumper at the end\"), call analyze_video first to identify the correct window, OR derive it from the storyboard timing already in the conversation (e.g. last beat's time range). Do not guess wildly — pick a sensible bumper/end-card window such as the final 1-3 seconds when the storyboard says scene_07 is 14-15s. Returns both the standalone replacement clip and the spliced composite. For LTX-2.3 and Wan 2.2 base videos the tool locks both ends with first/last-frame keyframes for seamless edges. For Seedance base videos the tool uses the original window as a reference for video-to-video transformation. If a requested window is shorter than the selected model's native render minimum, the handler renders a slightly larger handled clip, trims the result back to the requested seconds, then splices exactly that requested range. By default the regenerated segment's audio replaces the original audio in the [startSeconds, endSeconds] window, so new motion stays in sync with new sound. Pass keepOriginalAudio=true only when the user explicitly asks to keep the existing audio — phrasings like \"keep the audio\", \"leave the original audio\", \"preserve the music/score/dialogue\", \"don't change the audio\". If the user uses an ambiguous phrasing such as \"with the audio\" (which could mean either \"with the original audio kept\" or \"with new audio\"), DO NOT call this tool yet — first ask the user whether to preserve or replace the original audio in the replaced window. When replacementVideoIndex is set, the existing replacement clip's own audio is used; pass keepOriginalAudio=true only when the user explicitly wants the base video audio to stay over the replacement window.",
|
|
1001
1074
|
"parameters": {
|
|
1002
1075
|
"type": "object",
|
|
1003
1076
|
"properties": {
|
|
@@ -1035,12 +1108,15 @@
|
|
|
1035
1108
|
"type": "string",
|
|
1036
1109
|
"enum": [
|
|
1037
1110
|
"auto",
|
|
1111
|
+
"ltx25",
|
|
1038
1112
|
"ltx23",
|
|
1039
1113
|
"wan22",
|
|
1040
1114
|
"seedance2",
|
|
1041
|
-
"seedance2-
|
|
1115
|
+
"seedance2-mini",
|
|
1116
|
+
"seedance2-fast",
|
|
1117
|
+
"seedance2-5"
|
|
1042
1118
|
],
|
|
1043
|
-
"description": "Which model to use for the new segment. Default: \"auto\" —
|
|
1119
|
+
"description": "Which model to use for the new segment. Default: \"auto\" — preserve Seedance or WAN for matching base clips and otherwise use LTX 2.5. Use ltx23 only for explicit rollback. \"seedance2-5\" supports 480p/720p and 4-30s replacement windows at 24 fps."
|
|
1044
1120
|
},
|
|
1045
1121
|
"keepOriginalAudio": {
|
|
1046
1122
|
"type": "boolean",
|