@sogni-ai/sogni-intelligence-client 3.23.4 → 3.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/contracts/data/promptContracts.d.ts.map +1 -1
- package/dist/contracts/data/promptContracts.js +23 -10
- package/dist/contracts/data/promptContracts.js.map +1 -1
- package/dist/contracts/toolPromptMarkers.d.ts +5 -5
- package/dist/contracts/toolPromptMarkers.d.ts.map +1 -1
- package/dist/contracts/toolPromptMarkers.js +5 -5
- package/dist/contracts/toolPromptMarkers.js.map +1 -1
- package/dist/media/enhancementProfiles.d.ts +1 -1
- package/dist/media/enhancementProfiles.d.ts.map +1 -1
- package/dist/media/enhancementProfiles.js +0 -1
- package/dist/media/enhancementProfiles.js.map +1 -1
- package/dist/media/vendorModelPremium.d.ts +1 -1
- package/dist/media/vendorModelPremium.d.ts.map +1 -1
- package/dist/media/vendorModelPremium.js +2 -1
- package/dist/media/vendorModelPremium.js.map +1 -1
- package/dist/media/videoSettings.d.ts +20 -1
- package/dist/media/videoSettings.d.ts.map +1 -1
- package/dist/media/videoSettings.js +36 -1
- package/dist/media/videoSettings.js.map +1 -1
- package/dist/openai-tools/_manifests.generated.d.ts.map +1 -1
- package/dist/openai-tools/_manifests.generated.js +292 -73
- package/dist/openai-tools/_manifests.generated.js.map +1 -1
- package/dist/openai-tools/generation-tools.json +292 -73
- package/dist/public-skill-runtime/index.d.ts +19 -2
- package/dist/public-skill-runtime/index.d.ts.map +1 -1
- package/dist/public-skill-runtime/index.js +55 -2
- package/dist/public-skill-runtime/index.js.map +1 -1
- package/dist/schemas/tools/animate_photo.schema.json +59 -19
- package/dist/schemas/tools/generate_video.schema.json +66 -18
- package/dist/schemas/tools/sound_to_video.schema.json +41 -12
- package/dist/schemas/tools/video_to_video.schema.json +79 -13
- package/dist/skills/asset_reference_management/modelRefRegistry.d.ts.map +1 -1
- package/dist/skills/asset_reference_management/modelRefRegistry.js +25 -0
- package/dist/skills/asset_reference_management/modelRefRegistry.js.map +1 -1
- package/dist/skills/asset_reference_management/types.d.ts +1 -1
- package/dist/skills/asset_reference_management/types.d.ts.map +1 -1
- package/dist/tools/definitions/animate-photo/definition.d.ts.map +1 -1
- package/dist/tools/definitions/animate-photo/definition.js +23 -5
- package/dist/tools/definitions/animate-photo/definition.js.map +1 -1
- package/dist/tools/definitions/generate-video/definition.d.ts.map +1 -1
- package/dist/tools/definitions/generate-video/definition.js +32 -9
- package/dist/tools/definitions/generate-video/definition.js.map +1 -1
- package/dist/tools/definitions/sound-to-video/definition.d.ts.map +1 -1
- package/dist/tools/definitions/sound-to-video/definition.js +30 -5
- package/dist/tools/definitions/sound-to-video/definition.js.map +1 -1
- package/dist/tools/definitions/video-to-video/definition.d.ts.map +1 -1
- package/dist/tools/definitions/video-to-video/definition.js +43 -13
- package/dist/tools/definitions/video-to-video/definition.js.map +1 -1
- package/dist/tools/index.d.ts +2 -0
- package/dist/tools/index.d.ts.map +1 -1
- package/dist/tools/index.js +9 -2
- package/dist/tools/index.js.map +1 -1
- package/dist/tools/shared/modelRegistry.d.ts.map +1 -1
- package/dist/tools/shared/modelRegistry.js +4 -0
- package/dist/tools/shared/modelRegistry.js.map +1 -1
- package/dist/tools/shared/wan3References.d.ts +36 -0
- package/dist/tools/shared/wan3References.d.ts.map +1 -0
- package/dist/tools/shared/wan3References.js +78 -0
- package/dist/tools/shared/wan3References.js.map +1 -0
- package/dist/utils/videoModelIds.d.ts +2 -1
- package/dist/utils/videoModelIds.d.ts.map +1 -1
- package/dist/utils/videoModelIds.js +12 -0
- package/dist/utils/videoModelIds.js.map +1 -1
- package/dist-esm/contracts/data/promptContracts.js +23 -10
- package/dist-esm/contracts/data/promptContracts.js.map +1 -1
- package/dist-esm/contracts/toolPromptMarkers.js +5 -5
- package/dist-esm/contracts/toolPromptMarkers.js.map +1 -1
- package/dist-esm/media/enhancementProfiles.js +0 -1
- package/dist-esm/media/enhancementProfiles.js.map +1 -1
- package/dist-esm/media/vendorModelPremium.js +2 -1
- package/dist-esm/media/vendorModelPremium.js.map +1 -1
- package/dist-esm/media/videoSettings.js +35 -0
- package/dist-esm/media/videoSettings.js.map +1 -1
- package/dist-esm/openai-tools/_manifests.generated.js +292 -73
- package/dist-esm/openai-tools/_manifests.generated.js.map +1 -1
- package/dist-esm/openai-tools/generation-tools.json +292 -73
- package/dist-esm/public-skill-runtime/index.js +52 -1
- package/dist-esm/public-skill-runtime/index.js.map +1 -1
- package/dist-esm/schemas/tools/animate_photo.schema.json +59 -19
- package/dist-esm/schemas/tools/generate_video.schema.json +66 -18
- package/dist-esm/schemas/tools/sound_to_video.schema.json +41 -12
- package/dist-esm/schemas/tools/video_to_video.schema.json +79 -13
- package/dist-esm/skills/asset_reference_management/modelRefRegistry.js +25 -0
- package/dist-esm/skills/asset_reference_management/modelRefRegistry.js.map +1 -1
- package/dist-esm/tools/definitions/animate-photo/definition.js +23 -5
- package/dist-esm/tools/definitions/animate-photo/definition.js.map +1 -1
- package/dist-esm/tools/definitions/generate-video/definition.js +32 -9
- package/dist-esm/tools/definitions/generate-video/definition.js.map +1 -1
- package/dist-esm/tools/definitions/sound-to-video/definition.js +30 -5
- package/dist-esm/tools/definitions/sound-to-video/definition.js.map +1 -1
- package/dist-esm/tools/definitions/video-to-video/definition.js +43 -13
- package/dist-esm/tools/definitions/video-to-video/definition.js.map +1 -1
- package/dist-esm/tools/index.js +1 -0
- package/dist-esm/tools/index.js.map +1 -1
- package/dist-esm/tools/shared/modelRegistry.js +4 -0
- package/dist-esm/tools/shared/modelRegistry.js.map +1 -1
- package/dist-esm/tools/shared/wan3References.js +71 -0
- package/dist-esm/tools/shared/wan3References.js.map +1 -0
- package/dist-esm/utils/videoModelIds.js +11 -0
- package/dist-esm/utils/videoModelIds.js.map +1 -1
- package/package.json +3 -3
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
-
"$id": "https://schemas.sogni.ai/creative-agent/2026-
|
|
3
|
+
"$id": "https://schemas.sogni.ai/creative-agent/2026-07-18.1/tools/animate_photo.schema.json",
|
|
4
4
|
"title": "animate_photo arguments",
|
|
5
|
-
"schemaVersion": "2026-
|
|
6
|
-
"description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.5 by default, LTX 2.3 as rollback, or WAN 2.2. Do NOT use this tool for seedance2, seedance2-mini, or seedance2-5. Seedance media references — including Seedance 2.5 first-and-last-frame requests — must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.5 and LTX 2.3 generate audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Avoid placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, prefer generating both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, then call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. If the user explicitly wants separately created frame assets, preserve that staged instruction while keeping indices correct. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Call once you have clear creative intent. N-VIDEOS PATTERN: Avoid sequential animate_photo calls for N outputs. For a single fixed source/end frame where only prompt text varies, use sourceImageIndex + numberOfVariations=N + one Dynamic Prompt branch in prompt so Sogni submits one project with multiple jobs. If the user explicitly asks for Dynamic Prompt or Dynamic Template syntax, prefer this one-project path whenever every output uses the same source/end frames and shared settings, even if they also ask to stitch the completed clips afterward. Use sourceImageIndices/prompts for multi-segment stitched non-Seedance video, different source/end assets, different audio windows, different durations/dimensions, isolated retry lifecycle, or other per-output parameters. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips on ltx23 (30s total = 2 clips × 15s) and 10-second clips on wan22 (60s total on wan22 = 6 clips × 10s; do NOT pick 4 clips × 15s on wan22 — the wan22 worker rejects clips longer than 10s). Multi-source flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image and only prompt text varies, use sourceImageIndex=-1, frameRole=\"both\" if requested, endImageIndex=-1 if requested, numberOfVariations=N, and one Dynamic Prompt branch in prompt. For a long scripted/dialogue/storyboard video from a single supplied/uploaded image where each segment needs isolated exact dialogue or per-segment wiring, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when source/end asset wiring or other per-output parameters differ, pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame.",
|
|
5
|
+
"schemaVersion": "2026-07-18.1",
|
|
6
|
+
"description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.5 by default, LTX 2.3 as rollback, or WAN 2.2. Do NOT use this tool for seedance2, seedance2-mini, or seedance2-5. Seedance media references — including Seedance 2.5 first-and-last-frame requests — must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx25/ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.5 and LTX 2.3 generate audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Avoid placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN and MiniMax H3. For H3, use an I2V selector with frameRole=\"start\" for I2VA or frameRole=\"end\" for L2VA, and use an FLF2V selector with frameRole=\"both\" only when both endpoints are supplied. This H3 requirement overrides the later generic self-loop omission rule: even when the opening and closing frame are the same image, explicitly repeat that image in endImageIndex or endImageIndices — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, prefer generating both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, then call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. If the user explicitly wants separately created frame assets, preserve that staged instruction while keeping indices correct. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Call once you have clear creative intent. N-VIDEOS PATTERN: Avoid sequential animate_photo calls for N outputs. For a single fixed source/end frame where only prompt text varies, use sourceImageIndex + numberOfVariations=N + one Dynamic Prompt branch in prompt so Sogni submits one project with multiple jobs. If the user explicitly asks for Dynamic Prompt or Dynamic Template syntax, prefer this one-project path whenever every output uses the same source/end frames and shared settings, even if they also ask to stitch the completed clips afterward. Use sourceImageIndices/prompts for multi-segment stitched non-Seedance video, different source/end assets, different audio windows, different durations/dimensions, isolated retry lifecycle, or other per-output parameters. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips on ltx25 by default (or ltx23 rollback) (30s total = 2 clips × 15s) and 10-second clips on wan22 (60s total on wan22 = 6 clips × 10s; do NOT pick 4 clips × 15s on wan22 — the wan22 worker rejects clips longer than 10s). Multi-source flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image and only prompt text varies, use sourceImageIndex=-1, frameRole=\"both\" if requested, endImageIndex=-1 if requested, numberOfVariations=N, and one Dynamic Prompt branch in prompt. Each branch option must be a complete natural-language motion prompt; do not include \"clip N\", source-frame boilerplate, \"overall request context\", or instructions to follow the user request. For a long scripted/dialogue/storyboard video from a single supplied/uploaded image where each segment needs isolated exact dialogue or per-segment wiring, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when source/end asset wiring or other per-output parameters differ, pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame. Wan 3 first-frame and first+last-frame generation is supported with videoModel=\"wan3.0-video\".",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"properties": {
|
|
10
10
|
"prompt": {
|
|
11
11
|
"type": "string",
|
|
12
|
-
"description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.5 and LTX 2.3 generate audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same source/end frames and generation parameters and only prompt text varies. Example: \"{gentle sway with
|
|
12
|
+
"description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance or Wan 3 also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.5 and LTX 2.3 generate audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nPOSITIVE CONSTRAINT TRANSLATION: For LTX 2.3 and WAN 2.2, the prompt field is a positive prompt. Translate user avoid/no/don't constraints into affirmative production constraints instead of copying negative phrasing. Examples: \"no people in background\" -> single subject focus with an empty background; \"no text\" -> clean blank surfaces; \"don't make it blurry\" -> crisp sharp focus; \"no weird hands\" -> natural anatomically consistent hands; \"no mouth movement, no talking, no lip syncing\" -> silent expression-only physical performance with facial motion independent of speech timing; \"don't change the room\" -> the same room and layout remain consistent; \"keep flames consistent\" -> flame and ember movement remains consistent with the source scene. Preserve exact quoted visible text or dialogue when the user explicitly requests it, and keep surrounding surfaces blank. For Dynamic Prompt batches, put these translated shared constraints before the \"{...}\" branch so every variation inherits them.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements. Motion-only visual prompt; omit soundtrack, ambience, room tone, music, hums, sighs, spoken words, voice, and SFX cues because WAN does not generate audio.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same source/end frames and generation parameters and only prompt text varies. Example: \"{gentle sway with drifting embers|slow paw wave with a tiny head tilt|small hop with soft fur motion}\".\n\nMINIMAX H3 FRAME-ROLE PROMPTING: frameRole=\"start\" uses I2VA and describes motion forward from the supplied opening frame. frameRole=\"end\" uses L2VA: the supplied image is the closing frame, so infer a plausible earlier state and describe action, camera, objects, and scene gradually converging on it at the end; this overrides generic \"what happens next\" wording. frameRole=\"both\" uses FLF2VA and describes a coherent transition between the supplied opening and closing frames. The model-specific prompt shaper supplies and validates the exact official alignment line."
|
|
13
13
|
},
|
|
14
14
|
"expandPrompt": {
|
|
15
15
|
"type": "boolean",
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
},
|
|
18
18
|
"skipPromptProcessing": {
|
|
19
19
|
"type": "boolean",
|
|
20
|
-
"description": "Bypass automatic prompt shaping/refinement, image-description anchoring, transition-prompt rewriting, and voice-identity prompt formatting so the prompt text is sent unchanged to the video model. Set true ONLY when the user explicitly says not to modify/rewrite/enhance/expand/change/improve the prompt, or to use/send it exactly, verbatim, or as-is, AND the provided prompt already satisfies the tool requirements. Continue to set non-prompt parameters such as source indices, frameRole, model, duration, count, and aspect ratio. For Seedance literal prompt requests, also set expandPrompt=false. Do not set for ordinary underspecified requests."
|
|
20
|
+
"description": "Bypass automatic prompt shaping/refinement, image-description anchoring, transition-prompt rewriting, and voice-identity prompt formatting so the prompt text is sent unchanged to the video model. Set true ONLY when the user explicitly says not to modify/rewrite/enhance/expand/change/improve the prompt, or to use/send it exactly, verbatim, or as-is, AND the provided prompt already satisfies the tool requirements. Continue to set non-prompt parameters such as source indices, frameRole, model, duration, count, and aspect ratio. For Seedance or Wan 3 literal prompt requests, also set expandPrompt=false. Do not set for ordinary underspecified requests."
|
|
21
21
|
},
|
|
22
22
|
"videoModel": {
|
|
23
23
|
"type": "string",
|
|
@@ -30,29 +30,50 @@
|
|
|
30
30
|
"minimax-h3-i2v",
|
|
31
31
|
"minimax-h3-i2v-turbo",
|
|
32
32
|
"minimax-h3-flf2v",
|
|
33
|
-
"minimax-h3-flf2v-turbo"
|
|
33
|
+
"minimax-h3-flf2v-turbo",
|
|
34
|
+
"wan3.0-video"
|
|
34
35
|
],
|
|
35
|
-
"description": "Which video model to use. \"ltx25\" (default)
|
|
36
|
-
},
|
|
37
|
-
"generateAudio": {
|
|
38
|
-
"type": "boolean",
|
|
39
|
-
"description": "Whether to include generated/native audio for audio-capable models. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. When false, the returned video has no audio track. Ignored by audio-less WAN."
|
|
36
|
+
"description": "Which video model to use. \"ltx25\" (default) selects LTX 2.5 I2V/FLF with native audio; frameRole=\"both\" uses the FLF template with the I2V public model ID. Fast, HQ, and Pro currently use the release-validated official Distilled/Turbo workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe. \"ltx23\" remains an explicit rollback selector and is the only LTX family with the legacy transition/identity LoRAs. \"wan22\" is the fast WAN path without native audio. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\" and \"minimax-h3-i2v-turbo\": MiniMax H3 from exactly one endpoint image; use frameRole=\"start\" for an I2VA opening frame or frameRole=\"end\" for an L2VA closing frame. Both roles carry the sole endpoint in sourceImageIndex/sourceImageIndices; do not invent a minimax-h3-l2v selector and do not use endImageIndex/endImageIndices for last-frame-only L2VA. \"minimax-h3-flf2v\" and \"minimax-h3-flf2v-turbo\": MiniMax H3 first-and-last-frame interpolation; these require frameRole=\"both\" plus sourceImageIndex/sourceImageIndices for the opening frame and endImageIndex/endImageIndices for the closing frame. H3 renders 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio inside a 1344x768 pixel budget on a 32px grid. Do not set seedance2, seedance2-mini, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V, I2VA, L2VA, and FLF2VA prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music. I2VA prepends the official opening-frame alignment line, L2VA prepends the official duration-aware closing-frame alignment line, and FLF2VA prepends the official two-endpoint alignment line. For dialogue, use stable (S1) speaker IDs; keep identity, action, and delivery outside <d>, with only the language tag and exact spoken words inside <d>[Language] ...</d>. Use <scenetrans> at both connecting points when one line crosses a cut and explicitly state that its audio continues across the cut. Use the plain <cutoff> marker only when the video ending truncates speech; never emit tokenizer-internal <|...|> markers or plain caption/lyrics boundary tags. Do not merely delete pipe characters: caption markers become exact visible text in double quotes, lyrics markers become an ordinary <d>[Language] ...</d> singing block, and <|cutoff|> becomes plain <cutoff>. Standard uses 20 steps with res_multistep/simple. Turbo T2V, I2VA, L2VA, and FLF2VA use 4 steps with simple scheduling; er_sde is the default sampler, and direct CLI A/B overrides may select euler, er_sde, or sa_solver. Ref2VA Turbo is a separate 4-step Euler/simple workflow selected with minimax-h3-r2v-turbo; L2VA still uses the I2V selector with frameRole=\"end\". \"wan3.0-video\" is Alibaba Wan 3: one canonical premium-vendor model for text-to-video, first-frame and first+last-frame animation, loose multimodal references, audio-driven generation, and uploaded-video editing/extension. It renders 2-30s at fixed 30 fps with optional native audio, supports 480p/720p/1080p and 16:9/4:3/1:1/3:4/9:16, accepts up to 10 reference images, 5 reference videos, and 5 reference audios, and uses plain per-type prompt labels Image 1, Video 1, and Audio 1. Do not send negativePrompt. Use animate_photo for native first/last frames, generate_video for text or loose references, sound_to_video when audio is the primary driver, and video_to_video with controlMode=\"seedance-v2v\" for edits or extensions."
|
|
40
37
|
},
|
|
41
38
|
"negativePrompt": {
|
|
42
39
|
"type": "string",
|
|
43
|
-
"description": "Advanced LTX
|
|
40
|
+
"description": "Advanced LTX/WAN only. Use this field only when the user explicitly asks to set a separate negative prompt. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt.\n\nWan 3 has no negativePrompt request field; do not set this for wan3.0-video."
|
|
41
|
+
},
|
|
42
|
+
"generateAudio": {
|
|
43
|
+
"type": "boolean",
|
|
44
|
+
"description": "Whether the returned video should include generated/native audio. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. Supported by LTX and MiniMax H3; ignored by audio-less WAN.\n\nWan 3 supports this toggle; omit it for audio-on by default or set false only for an explicitly silent result."
|
|
44
45
|
},
|
|
45
46
|
"duration": {
|
|
46
47
|
"type": "number",
|
|
47
|
-
"description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Per-model maximum: ltx25 and ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip."
|
|
48
|
+
"description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Per-model maximum: ltx25 and ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), wan3.0-video = 30s with a 2s minimum, minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip."
|
|
49
|
+
},
|
|
50
|
+
"smartDuration": {
|
|
51
|
+
"type": "boolean",
|
|
52
|
+
"description": "Wan 3 only. Let the model choose 2-30 seconds. Do not also set duration. The 30-second maximum is reserved and the final charge settles down to reported duration."
|
|
53
|
+
},
|
|
54
|
+
"ratio": {
|
|
55
|
+
"type": "string",
|
|
56
|
+
"enum": [
|
|
57
|
+
"adaptive",
|
|
58
|
+
"16:9",
|
|
59
|
+
"4:3",
|
|
60
|
+
"1:1",
|
|
61
|
+
"3:4",
|
|
62
|
+
"9:16"
|
|
63
|
+
],
|
|
64
|
+
"description": "Wan 3 only. Use \"adaptive\" to preserve the source frame shape."
|
|
65
|
+
},
|
|
66
|
+
"watermark": {
|
|
67
|
+
"type": "boolean",
|
|
68
|
+
"description": "Wan 3 only. Add Alibaba's visible watermark. Defaults to false."
|
|
48
69
|
},
|
|
49
70
|
"targetResolution": {
|
|
50
71
|
"type": "number",
|
|
51
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\",
|
|
72
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"1080p\" without exact pixels or an output orientation. This preserves the source image aspect ratio. Wan 3 supports 480p, 720p, and 1080p; HappyHorse supports only 720p and 1080p. Never set 4K for either. MiniMax H3 renders inside a 1344x768 pixel budget on a 32px grid, so use 768 for H3 and never 1080p or 4K. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\" or \"720p landscape\", use exact-pixel aspectRatio instead."
|
|
52
73
|
},
|
|
53
74
|
"sourceImageIndex": {
|
|
54
75
|
"type": "number",
|
|
55
|
-
"description": "Which image to use as the START frame. Use 0-based non-negative indices for generated result images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Omit to auto-select: uses the latest result for \"start\"/\"end\" modes, or the FIRST result for \"both\" mode. IMPORTANT: When frameRole is \"both\", set this to the start frame image index and endImageIndex to the end frame image index."
|
|
76
|
+
"description": "Which image to use as the START frame. Use 0-based non-negative indices for generated result images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Omit to auto-select: uses the latest result for \"start\"/\"end\" modes, or the FIRST result for \"both\" mode. IMPORTANT: When frameRole is \"both\", set this to the start frame image index and endImageIndex to the end frame image index.\n\nMiniMax H3: with an I2V selector this field is the sole endpoint image. frameRole=\"start\" treats it as the I2VA opening frame; frameRole=\"end\" treats it as the L2VA closing frame. With an FLF2V selector and frameRole=\"both\", this field is the required opening frame and endImageIndex is the required closing frame. For a same-image H3 loop, explicitly repeat this index in endImageIndex instead of omitting the closing field."
|
|
56
77
|
},
|
|
57
78
|
"sourceImageIndices": {
|
|
58
79
|
"type": "array",
|
|
@@ -61,7 +82,7 @@
|
|
|
61
82
|
},
|
|
62
83
|
"minItems": 1,
|
|
63
84
|
"maxItems": 16,
|
|
64
|
-
"description": "Array of source frame indices — one video is generated per entry as its own SDK project, all running in PARALLEL. Use this when outcomes need different source images, different end frames, isolated retry lifecycle, or other per-clip asset wiring/parameters. If every outcome uses the same source/end frames and only prompt text differs, prefer sourceImageIndex with numberOfVariations=N and one Dynamic Prompt branch in `prompt` so Sogni creates one project with multiple jobs. Use 0-based non-negative result indices for generated images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Repeating -1 is allowed for true multi-project workflows that intentionally reuse the same uploaded image while varying per-clip assets or parameters. By default all projects share the `prompt`/`voice`/`duration`, but you can pass `prompts` (array) to give each clip its own dialogue/motion when multi-project fan-out is required. Avoid sequential animate_photo calls for N outputs. Do NOT combine with `numberOfVariations` or `sourceImageIndex`. Use frameRole=\"end\" with sourceImageIndices only when the user explicitly says the repeated uploaded/generated image is the last/end frame for each clip and no first/start frame should be supplied; in that case omit endImageIndex/endImageIndices because each sourceImageIndices entry is the end frame. You MAY combine with frameRole=\"both\" when clips need start and end frames. For adjacent transition chains across generated images, use sourceImageIndices=[start..end-1] and endImageIndices=[start+1..end] so N images produce N-1 transition clips. If the uploaded/original image starts the chain and generated results are the remaining frames, use sourceImageIndices=[-1,start..end-1] and endImageIndices=[start..end]. If the user supplies multiple uploaded images as the actual keyframe sequence, use adjacent negative uploaded indices, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4], endImageIndices=[-2,-3,-4,-5], frameRole=\"both\", prompts length 4, then stitch_video. If the user specifies transition motion, camera behavior, actions, dialogue, or audio, copy those instructions into every corresponding per-clip prompt; only invent a generic smooth transition when the user does not specify one. If the user asks for a seamless loop or final transition from the last image back to the first, close the chain by including the last image as a source and the first image as the final end frame, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4,-5], endImageIndices=[-2,-3,-4,-5,-1]. For generated scene keyframes that should each loop to themselves, omit endImageIndex/endImageIndices so each source image is also its own end frame. Set endImageIndex=-1 only when every sourceImageIndices entry is also -1 and every segment reuses the first uploaded image. Range: 1–16 indices. For generated image batches, values MUST be read from the latest edit_image/generate_image tool result's `startIndex` field. If startIndex=3 and 4 images were generated in that batch, pass `[3,4,5,6]` (NOT `[0,1,2,3]`). Do NOT assume generated indices start at 0 — they don't if there are prior results in the conversation."
|
|
85
|
+
"description": "Array of source frame indices — one video is generated per entry as its own SDK project, all running in PARALLEL. Use this when outcomes need different source images, different end frames, isolated retry lifecycle, or other per-clip asset wiring/parameters. If every outcome uses the same source/end frames and only prompt text differs, prefer sourceImageIndex with numberOfVariations=N and one Dynamic Prompt branch in `prompt` so Sogni creates one project with multiple jobs. Use 0-based non-negative result indices for generated images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Repeating -1 is allowed for true multi-project workflows that intentionally reuse the same uploaded image while varying per-clip assets or parameters. By default all projects share the `prompt`/`voice`/`duration`, but you can pass `prompts` (array) to give each clip its own dialogue/motion when multi-project fan-out is required. Avoid sequential animate_photo calls for N outputs. Do NOT combine with `numberOfVariations` or `sourceImageIndex`. Use frameRole=\"end\" with sourceImageIndices only when the user explicitly says the repeated uploaded/generated image is the last/end frame for each clip and no first/start frame should be supplied; in that case omit endImageIndex/endImageIndices because each sourceImageIndices entry is the end frame. You MAY combine with frameRole=\"both\" when clips need start and end frames. For adjacent transition chains across generated images, use sourceImageIndices=[start..end-1] and endImageIndices=[start+1..end] so N images produce N-1 transition clips. If the uploaded/original image starts the chain and generated results are the remaining frames, use sourceImageIndices=[-1,start..end-1] and endImageIndices=[start..end]. If the user supplies multiple uploaded images as the actual keyframe sequence, use adjacent negative uploaded indices, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4], endImageIndices=[-2,-3,-4,-5], frameRole=\"both\", prompts length 4, then stitch_video. If the user specifies transition motion, camera behavior, actions, dialogue, or audio, copy those instructions into every corresponding per-clip prompt; only invent a generic smooth transition when the user does not specify one. If the user asks for a seamless loop or final transition from the last image back to the first, close the chain by including the last image as a source and the first image as the final end frame, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4,-5], endImageIndices=[-2,-3,-4,-5,-1]. For generated scene keyframes that should each loop to themselves, omit endImageIndex/endImageIndices so each source image is also its own end frame. Set endImageIndex=-1 only when every sourceImageIndices entry is also -1 and every segment reuses the first uploaded image. Range: 1–16 indices. For generated image batches, values MUST be read from the latest edit_image/generate_image tool result's `startIndex` field. If startIndex=3 and 4 images were generated in that batch, pass `[3,4,5,6]` (NOT `[0,1,2,3]`). Do NOT assume generated indices start at 0 — they don't if there are prior results in the conversation.\n\nMiniMax H3 fan-out: with an I2V selector, every entry is an opening frame when frameRole=\"start\" or a closing frame when frameRole=\"end\". With an FLF2V selector and frameRole=\"both\", every entry is an opening frame and uses an allowed shared endImageIndex or a corresponding endImageIndices entry under the existing fan-out rules. H3 overrides the generic self-loop omission rule: same-image loops must set endImageIndices equal to sourceImageIndices, or use an allowed shared endImageIndex where the existing fan-out rules permit it."
|
|
65
86
|
},
|
|
66
87
|
"prompts": {
|
|
67
88
|
"type": "array",
|
|
@@ -89,11 +110,11 @@
|
|
|
89
110
|
"end",
|
|
90
111
|
"both"
|
|
91
112
|
],
|
|
92
|
-
"description": "How to use the source image(s)
|
|
113
|
+
"description": "How to use the source image(s). \"start\" (default): first frame. \"end\": last frame. \"both\": interpolate between first and last frames. For MiniMax H3, use minimax-h3-i2v or minimax-h3-i2v-turbo with frameRole=\"start\" for I2VA or frameRole=\"end\" for L2VA; sourceImageIndex/sourceImageIndices carries the sole endpoint for either mode. Do not invent a minimax-h3-l2v selector and do not set endImageIndex/endImageIndices for last-frame-only L2VA. MiniMax H3 first-and-last-frame interpolation uses minimax-h3-flf2v or minimax-h3-flf2v-turbo and requires frameRole=\"both\" plus both source and end image fields, even when both fields repeat the same image for a loop. This explicit H3 closing-field requirement overrides generic self-loop guidance that permits omitting an end field. For single non-H3 clips using \"both\", set sourceImageIndex and endImageIndex; fan-out can use matching sourceImageIndices/endImageIndices."
|
|
93
114
|
},
|
|
94
115
|
"endImageIndex": {
|
|
95
116
|
"type": "number",
|
|
96
|
-
"description": "Which image to use as the END frame. Use 0-based non-negative indices for generated results. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. For a single frameRole=\"both\" transition between two different images, set this to the desired end frame. For sourceImageIndices fan-out where each generated keyframe should also be its own last frame, OMIT this field. Use a shared uploaded endImageIndex only when every sourceImageIndices entry is also an uploaded image; otherwise use endImageIndices for per-clip end frames."
|
|
117
|
+
"description": "Which image to use as the END frame. Use 0-based non-negative indices for generated results. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. For a single frameRole=\"both\" transition between two different images, set this to the desired end frame. For sourceImageIndices fan-out where each generated keyframe should also be its own last frame, OMIT this field. Use a shared uploaded endImageIndex only when every sourceImageIndices entry is also an uploaded image; otherwise use endImageIndices for per-clip end frames.\n\nMiniMax H3: use this only with an FLF2V selector and frameRole=\"both\" as the closing frame, including supported shared-end fan-out. Do not omit it for a single same-image H3 loop; repeat sourceImageIndex here. For last-frame-only L2VA, use the I2V selector with frameRole=\"end\" and carry the sole closing frame in sourceImageIndex instead."
|
|
97
118
|
},
|
|
98
119
|
"endImageIndices": {
|
|
99
120
|
"type": "array",
|
|
@@ -102,11 +123,30 @@
|
|
|
102
123
|
},
|
|
103
124
|
"minItems": 1,
|
|
104
125
|
"maxItems": 16,
|
|
105
|
-
"description": "Per-clip END frame indices for sourceImageIndices fan-out. Use ONLY with frameRole=\"both\". Length MUST exactly match sourceImageIndices. Use 0-based non-negative indices for generated results and negative indices for uploaded images (-1 first upload, -2 second upload, etc.). Use this for transition chains between generated images, e.g. 5 generated images at indices [0,1,2,3,4] should become 4 transition clips with sourceImageIndices=[0,1,2,3], endImageIndices=[1,2,3,4], prompts length 4, duration as requested, then stitch_video. If the chain starts on the uploaded image and continues through generated results [0,1,2,3], use sourceImageIndices=[-1,0,1,2] and endImageIndices=[0,1,2,3]. If the user supplies 5 uploaded images as the sequence, use sourceImageIndices=[-1,-2,-3,-4] and endImageIndices=[-2,-3,-4,-5]. If the user requests a seamless loop or final transition back to the first image, append that loop closure: sourceImageIndices=[-1,-2,-3,-4,-5], endImageIndices=[-2,-3,-4,-5,-1]. Do NOT also set endImageIndex when using this."
|
|
126
|
+
"description": "Per-clip END frame indices for sourceImageIndices fan-out. Use ONLY with frameRole=\"both\". Length MUST exactly match sourceImageIndices. Use 0-based non-negative indices for generated results and negative indices for uploaded images (-1 first upload, -2 second upload, etc.). Use this for transition chains between generated images, e.g. 5 generated images at indices [0,1,2,3,4] should become 4 transition clips with sourceImageIndices=[0,1,2,3], endImageIndices=[1,2,3,4], prompts length 4, duration as requested, then stitch_video. If the chain starts on the uploaded image and continues through generated results [0,1,2,3], use sourceImageIndices=[-1,0,1,2] and endImageIndices=[0,1,2,3]. If the user supplies 5 uploaded images as the sequence, use sourceImageIndices=[-1,-2,-3,-4] and endImageIndices=[-2,-3,-4,-5]. If the user requests a seamless loop or final transition back to the first image, append that loop closure: sourceImageIndices=[-1,-2,-3,-4,-5], endImageIndices=[-2,-3,-4,-5,-1]. Do NOT also set endImageIndex when using this.\n\nMiniMax H3 fan-out: use this only with an FLF2V selector and frameRole=\"both\", paired one-to-one with sourceImageIndices. For same-image H3 loops, set this array equal to sourceImageIndices instead of omitting it. For last-frame-only L2VA fan-out, use the I2V selector with frameRole=\"end\" and carry closing frames in sourceImageIndices instead."
|
|
106
127
|
},
|
|
107
128
|
"voicePersonaName": {
|
|
108
129
|
"type": "string",
|
|
109
130
|
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
|
|
131
|
+
},
|
|
132
|
+
"loras": {
|
|
133
|
+
"type": "array",
|
|
134
|
+
"minItems": 1,
|
|
135
|
+
"maxItems": 8,
|
|
136
|
+
"items": {
|
|
137
|
+
"type": "string",
|
|
138
|
+
"minLength": 1
|
|
139
|
+
},
|
|
140
|
+
"description": "Ordered LoRA IDs to apply to a MiniMax H3 render. Use only when the user explicitly asks for a LoRA or for an effect one of these names describes. Stack up to 8 in one request; order matters because the adapters apply in sequence and do not commute. Keep this array positionally aligned with loraStrengths. The first render with an uncached LoRA takes longer to start while the worker downloads it.\n\nAccepted only when videoModel is one of \"minimax-h3-i2v\", \"minimax-h3-i2v-turbo\", \"minimax-h3-flf2v\", \"minimax-h3-flf2v-turbo\". Every other video model on this tool loads no LoRAs and silently ignores these arrays, so set videoModel to an H3 mode in the same call when the user asks for one.\n\nOne LoRA is published for MiniMax H3 today: h3-realism-people (fal), a realism pass trained on live-action footage of people. It restores skin texture and pores, stray hairs, fabric weave and a fine sensor grain that the base model smooths away, and holds up in close-up. It needs its trigger word: put r34l1sm near the FRONT of the prompt, or the render comes back as ordinary H3 with no error. Exact ranges and any LoRA published since: GET /v1/loras/comfy?modelId=<model>. Do not invent ids."
|
|
141
|
+
},
|
|
142
|
+
"loraStrengths": {
|
|
143
|
+
"type": "array",
|
|
144
|
+
"minItems": 1,
|
|
145
|
+
"maxItems": 8,
|
|
146
|
+
"items": {
|
|
147
|
+
"type": "number"
|
|
148
|
+
},
|
|
149
|
+
"description": "Strength for each LoRA in loras, in the same order. Omitting the array applies 1.0 to every LoRA, which is NOT the catalog default and for h3-realism-people is already at the top of its band, so send explicit values. Video LoRAs are positive-only — unlike the bipolar Krea 2 image sliders, a negative value is not an inverse effect and 0 is off. h3-realism-people takes 0-2 and its catalog default is 0.8; 0.6-1 is the usable band. It also pulls the camera in as it climbs: at 1.5 and above the shot reliably recomposes and the grade darkens, which on an image-conditioned mode can crop the subject out of the frame the user supplied. Raise it above 1 only when the user asks for more, and prefer the default when they supplied a first or last frame."
|
|
110
150
|
}
|
|
111
151
|
},
|
|
112
152
|
"required": [
|
|
@@ -1,33 +1,61 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
-
"$id": "https://schemas.sogni.ai/creative-agent/2026-
|
|
3
|
+
"$id": "https://schemas.sogni.ai/creative-agent/2026-07-18.1/tools/generate_video.schema.json",
|
|
4
4
|
"title": "generate_video arguments",
|
|
5
|
-
"schemaVersion": "2026-
|
|
6
|
-
"description": "Generate a video from text or Seedance multimodal references. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as
|
|
5
|
+
"schemaVersion": "2026-07-18.1",
|
|
6
|
+
"description": "Generate a video from text or Seedance multimodal references. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as rollback and also generates audio natively (dialogue, sounds, ambient music) — describe audio in the prompt. If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. For non-Seedance syncing to a specific song or audio track, use sound_to_video instead. For non-Seedance animation from a locked source photo, use animate_photo. Do NOT use for My Personas unless generating a Seedance reference-based video — standard persona videos use resolve_personas → edit_image → animate_photo. SEEDANCE DEFAULT: For seedance2, seedance2-mini, or seedance2-5, default to exactly one video (4-15s on 2.0/Mini, 4-30s on seedance2-5) unless the user explicitly asks for multiple separate outputs. Multiple beats, shots, or scene descriptions in one up-to-15s Seedance prompt are still one video. If the user requests one continuous Seedance video longer than 15s, prefer seedance2-5, which renders up to 30s in a single call; beyond 30s (or on 2.0/Mini) preserve the requested total duration in the prompt/context and let chat orchestration split it into supported segment renders and stitch them instead of clamping it to a short excerpt. Uploaded/generated storyboard, shot-sheet, or trailer-concept images used as Seedance references should become one Seedance generate_video call by default; do not extract panels with edit_image and do not animate the storyboard sheet with LTX unless the user explicitly asks for separate non-Seedance clips. Seedance loose image, video, and audio references go through this tool; do not use animate_photo sourceImageIndex/frameRole/endImageIndex for Seedance. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. If the uploaded audio is the primary sync target, lip-sync target, or requested as sound-to-video/audio-sync, use sound_to_video with videoModel=\"seedance2-mini\" instead of this tool unless the user asks for full Seedance. Use referenceAudioIndices here only when audio is a loose reference under an image/video-anchored Seedance shot. For Seedance, every image — first frame, last frame, or loose reference — is passed through referenceImageIndices (auto-uploaded as referenceImageUrls). Anchor frame intent in the prompt with @Image tags such as \"Use @Image1 as the opening shot reference. Begin the video with a composition, subject placement, lighting, mood, and camera framing that closely match @Image1.\" (or @Image2 as the final shot reference). For seamless-loop or \"first frame and last frame identical\" requests with a single uploaded image, anchor it explicitly as both: \"Use @Image1 as both the first frame and last frame so the video loops cleanly back to the opening composition.\" Assign each useful @Image/@Video/@Audio tag a role. APPROVED STORYBOARD PRODUCTION: When the user asks for a production workflow from an approved storyboard, the chat orchestrator should use the durable CampaignStoryboard contract: render the composite board, audit it, generate per-scene GPT Image 2 keyframes, then render Seedance scene clips and stitch them. Do not replace that with a generic storyboard-reference video unless the user asks for a fast draft. PARTIAL VIDEO EDITS: Do NOT call generate_video to re-render an existing rendered/uploaded video just to change part of it (the bumper, the intro, the end card, a single scene, the last few seconds, etc.). Use replace_video_segment for that — it preserves the unchanged portion, keeps the original audio outside the replaced window, and costs far less. Likewise use extend_video to add new time to the end without rewriting the rest. If the request is vague, ask about vision/mood/style first. Only call once you have clear creative intent. WAN 3 uses the exact selector wan3.0-video: use this tool for text-to-video or loose Image 1/Video 1/Audio 1 references; use animate_photo for native first/last frames, sound_to_video when audio drives timing, and video_to_video for source-video edits.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"properties": {
|
|
10
10
|
"prompt": {
|
|
11
11
|
"type": "string",
|
|
12
|
-
"description": "Write one flowing paragraph like a cinematographer describing a shot. Present tense, specific natural language. Longer clips need longer prompts; close-ups need more detail than wide shots.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: shot/style → subject (age, clothing, hairstyle, distinguishing details) → environment, lighting, atmosphere → action beat by beat → camera movement → audio and dialogue.\n\nCAST CONTINUITY: For screenplay, script, storyboard, commercial, series, or other longer-form video tasks with recurring characters, use stable character names and repeat the same visual anchors every time they appear (age range, build, hairstyle, outfit silhouette, color palette, signature prop/accessory, posture, voice). Do not rename, merge, redesign, or drift characters between scenes unless the user asks.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Direct the layout like scene blocking. State left/right placement, foreground/background, facing toward/away, and relative distance when multiple subjects or important objects are involved.\n\nACTION: Drive motion with concrete verbs. Specify who moves, what moves, how it moves, and what the camera does. Avoid generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior — not \"she is sad\", instead \"she looks down, pauses, and her voice cracks\". QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nSTORYBOARD TEXT: For storyboard references, structural headings, section numbers, slide titles, panel titles, and captions may become short audio-only narration/voiceover or key-message beats, but they are not subtitles, title cards, lower thirds, or visible overlays unless the user explicitly asks for visible text/on-screen text/title card/subtitle/lower third/signage/CTA. Do not concatenate storyboard labels into run-on voiceover; use separate brief phrases with pauses.\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — close-up, tracking shot, dolly in, handheld, slow arc, static frame. Describe movement relative to subject.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters, dialogue, skits): capture the full creative intent. The system auto-expands into a detailed prompt.\n\nAVOID: Vague prompts, too many characters at once, conflicting lighting logic, readable text or logos, abstract emotions with no visible behavior, rigid numeric constraints (exact angles, counts, speeds).\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax. Lock in any camera/subject/style the user specified, vary the rest. Example: \"slow dolly in on a city street {at dawn with golden light|during a rainstorm|at night with neon reflections}\"."
|
|
12
|
+
"description": "Write one flowing paragraph like a cinematographer describing a shot. Present tense, specific natural language. Longer clips need longer prompts; close-ups need more detail than wide shots.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance or Wan 3 also set expandPrompt=false.\n\nSTRUCTURE: shot/style → subject (age, clothing, hairstyle, distinguishing details) → environment, lighting, atmosphere → action beat by beat → camera movement → audio and dialogue.\n\nCAST CONTINUITY: For screenplay, script, storyboard, commercial, series, or other longer-form video tasks with recurring characters, use stable character names and repeat the same visual anchors every time they appear (age range, build, hairstyle, outfit silhouette, color palette, signature prop/accessory, posture, voice). Do not rename, merge, redesign, or drift characters between scenes unless the user asks.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Direct the layout like scene blocking. State left/right placement, foreground/background, facing toward/away, and relative distance when multiple subjects or important objects are involved.\n\nACTION: Drive motion with concrete verbs. Specify who moves, what moves, how it moves, and what the camera does. Avoid generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior — not \"she is sad\", instead \"she looks down, pauses, and her voice cracks\". QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nSTORYBOARD TEXT: For storyboard references, structural headings, section numbers, slide titles, panel titles, and captions may become short audio-only narration/voiceover or key-message beats, but they are not subtitles, title cards, lower thirds, or visible overlays unless the user explicitly asks for visible text/on-screen text/title card/subtitle/lower third/signage/CTA. Do not concatenate storyboard labels into run-on voiceover; use separate brief phrases with pauses.\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — close-up, tracking shot, dolly in, handheld, slow arc, static frame. Describe movement relative to subject.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters, dialogue, skits): capture the full creative intent. The system auto-expands into a detailed prompt.\n\nAVOID: Vague prompts, too many characters at once, conflicting lighting logic, readable text or logos, abstract emotions with no visible behavior, rigid numeric constraints (exact angles, counts, speeds).\n\nNON-SEEDANCE POSITIVE CONSTRAINTS: For videoModel=\"ltx25\", \"ltx23\", or \"wan22\", prompt is a positive prompt. Translate user avoid/no/don't constraints into affirmative production constraints instead of copying negative phrasing. Preserve exact quoted visible text or dialogue when the user explicitly requests it; keep surrounding surfaces blank.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same references, model, duration, dimensions, and generation parameters and only prompt text varies. Lock in any camera/subject/style the user specified, vary the rest. Example: \"slow dolly in on a city street {at dawn with golden light|during a rainstorm|at night with neon reflections}\"."
|
|
13
13
|
},
|
|
14
14
|
"expandPrompt": {
|
|
15
15
|
"type": "boolean",
|
|
16
|
-
"description": "Seedance only. Whether to
|
|
16
|
+
"description": "Seedance and Wan 3 only. Whether to expand the prompt before dispatch. Defaults to true. For Wan 3, a successful Sogni expansion disables Alibaba prompt_extend to prevent a second rewrite; false disables both expansion layers so exact prompts remain exact."
|
|
17
17
|
},
|
|
18
18
|
"skipPromptProcessing": {
|
|
19
19
|
"type": "boolean",
|
|
20
|
-
"description": "Bypass automatic prompt shaping/refinement and voice-identity prompt formatting so the prompt text is sent unchanged to the video model. Set true ONLY when the user explicitly says not to modify/rewrite/enhance/expand/change/improve the prompt, or to use/send it exactly, verbatim, or as-is, AND the provided prompt already satisfies the tool requirements. Continue to set non-prompt parameters such as model, duration, count, aspect ratio, and seed. For Seedance literal prompt requests, also set expandPrompt=false. Do not set for ordinary underspecified requests."
|
|
20
|
+
"description": "Bypass automatic prompt shaping/refinement and voice-identity prompt formatting so the prompt text is sent unchanged to the video model. Set true ONLY when the user explicitly says not to modify/rewrite/enhance/expand/change/improve the prompt, or to use/send it exactly, verbatim, or as-is, AND the provided prompt already satisfies the tool requirements. Continue to set non-prompt parameters such as model, duration, count, aspect ratio, and seed. For Seedance or Wan 3 literal prompt requests, also set expandPrompt=false. Do not set for ordinary underspecified requests."
|
|
21
21
|
},
|
|
22
22
|
"duration": {
|
|
23
23
|
"type": "number",
|
|
24
|
-
"description": "Video duration in seconds. Default: 5.
|
|
24
|
+
"description": "Video duration in seconds. Default: 5. Per-model range: LTX 2.3 and LTX 2.5 = 2-20s; Wan 3 = 2-30s; Seedance 2.0 and Mini = 4-15s; Seedance 2.5 = 4-30s; HappyHorse 1.1 = 3-15s. Use when the user explicitly requests a specific length. MiniMax H3 is quantized to a 17-frame grid at a fixed 24 fps and renders 124-362 frames, so an H3 clip runs 5.17-15.08 seconds and a requested length outside that window snaps to the nearest valid H3 length.",
|
|
25
25
|
"minimum": 2,
|
|
26
26
|
"maximum": 30
|
|
27
27
|
},
|
|
28
|
+
"smartDuration": {
|
|
29
|
+
"type": "boolean",
|
|
30
|
+
"description": "Wan 3 only. Set true to let Wan 3 choose an output length from 2-30 seconds. Do not also set duration. Sogni reserves the 30-second maximum before generation and settles the completed job down to Alibaba's reported output duration."
|
|
31
|
+
},
|
|
32
|
+
"ratio": {
|
|
33
|
+
"type": "string",
|
|
34
|
+
"enum": [
|
|
35
|
+
"adaptive",
|
|
36
|
+
"16:9",
|
|
37
|
+
"4:3",
|
|
38
|
+
"1:1",
|
|
39
|
+
"3:4",
|
|
40
|
+
"9:16"
|
|
41
|
+
],
|
|
42
|
+
"description": "Wan 3 only. Output ratio. Use \"adaptive\" to inherit source/context shape; extension requires adaptive. Omit for automatic behavior."
|
|
43
|
+
},
|
|
44
|
+
"watermark": {
|
|
45
|
+
"type": "boolean",
|
|
46
|
+
"description": "Wan 3 only. Add Alibaba's visible watermark. Defaults to false."
|
|
47
|
+
},
|
|
48
|
+
"referenceFileUrl": {
|
|
49
|
+
"type": "string",
|
|
50
|
+
"description": "Wan 3 only. One public HTTPS document URL for context (DOCX/DOC/XLSX/XLS/PPTX/PPT/PDF/TXT/KEY/PAGES/NUMBERS/Markdown, up to 100 MB; PDF/DOCX/DOC/PPTX/PPT/KEY/PAGES up to 50 pages). Mutually exclusive with referenceLinkUrl and first/last-frame inputs."
|
|
51
|
+
},
|
|
52
|
+
"referenceLinkUrl": {
|
|
53
|
+
"type": "string",
|
|
54
|
+
"description": "Wan 3 only. One public HTTPS webpage URL for context. Mutually exclusive with referenceFileUrl and first/last-frame inputs."
|
|
55
|
+
},
|
|
28
56
|
"negativePrompt": {
|
|
29
57
|
"type": "string",
|
|
30
|
-
"description": "Advanced LTX
|
|
58
|
+
"description": "Advanced LTX/WAN only. Use this field only when the user explicitly asks to set a separate negative prompt. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt. Do not set for MiniMax H3, Seedance, or HappyHorse.\n\nWan 3 has no negativePrompt request field; do not set this for wan3.0-video."
|
|
31
59
|
},
|
|
32
60
|
"videoModel": {
|
|
33
61
|
"type": "string",
|
|
@@ -44,50 +72,51 @@
|
|
|
44
72
|
"happyhorse-1.1-i2v",
|
|
45
73
|
"happyhorse-1.1-r2v",
|
|
46
74
|
"minimax-h3-r2v",
|
|
47
|
-
"minimax-h3-r2v-turbo"
|
|
75
|
+
"minimax-h3-r2v-turbo",
|
|
76
|
+
"wan3.0-video"
|
|
48
77
|
],
|
|
49
|
-
"description": "Video model. \"ltx25\" (default)
|
|
78
|
+
"description": "Video model. \"ltx25\" (default) selects LTX 2.5 standard generation with native audio. Fast, HQ, and Pro currently use the release-validated official Distilled/Turbo workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe. \"ltx23\" remains available as an explicit rollback selector. \"wan22\" is the fast WAN path without native audio. LTX 2.5 V2V controls are exposed separately through video_to_video; voice ID-LoRA, transition LoRA, and 10Eros remain LTX 2.3-only. HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 text-to-video uses \"minimax-h3-t2v\". H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\", a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference (image or video) is required; audio alone is invalid. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" or \"minimax-h3-i2v-turbo\" and frameRole=\"start\" for an opening-frame I2VA animation, or frameRole=\"end\" for a closing-frame L2VA animation. Use \"minimax-h3-flf2v\" or \"minimax-h3-flf2v-turbo\" with frameRole=\"both\" only for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft, Mini, or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, first-and-last-frame conditioning, and a much larger reference budget than the 2.0 family: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Choose \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s in one call instead of being split and stitched), or wants a first-and-last-frame Seedance transition. Keep \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance supports multimodal loose reference assets. Seedance 2.0 and Mini accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with up to 50 reference media files total, subject to those per-modality caps. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. MiniMax H3 Base and Turbo T2V, I2VA, L2VA, and FLF2VA prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music. I2VA prepends the official opening-frame alignment line, L2VA prepends the official duration-aware closing-frame alignment line, and FLF2VA prepends the official two-endpoint alignment line. For dialogue, use stable (S1) speaker IDs; keep identity, action, and delivery outside <d>, with only the language tag and exact spoken words inside <d>[Language] ...</d>. Use <scenetrans> at both connecting points when one line crosses a cut and explicitly state that its audio continues across the cut. Use the plain <cutoff> marker only when the video ending truncates speech; never emit tokenizer-internal <|...|> markers or plain caption/lyrics boundary tags. Do not merely delete pipe characters: caption markers become exact visible text in double quotes, lyrics markers become an ordinary <d>[Language] ...</d> singing block, and <|cutoff|> becomes plain <cutoff>. Standard uses 20 steps with res_multistep/simple. Turbo T2V, I2VA, L2VA, and FLF2VA use 4 steps with simple scheduling; er_sde is the default sampler, and direct CLI A/B overrides may select euler, er_sde, or sa_solver. Ref2VA Turbo is a separate 4-step Euler/simple workflow selected with minimax-h3-r2v-turbo; L2VA still uses the I2V selector with frameRole=\"end\". MiniMax H3 R2V requires at least one visual reference (image or video); audio alone is invalid. Its six ordered prompt sections are subject_definitions, summary, retention_analysis, detailed_description, overall_soundscape, and non_diegetic_music. Use <Subject N> for reusable visible content, <Picture N> only for concrete keyframes/composition anchors, <Video N> for whole-video relationships, and <Audio N> for copied or referenced audio. Use minimax-h3-r2v for the standard 20-step Ref2VA workflow and minimax-h3-r2v-turbo for the dedicated LightX2V 4-step Euler/simple Turbo workflow with its upstream-aligned 960x544 default. \"wan3.0-video\" is Alibaba Wan 3: one canonical premium-vendor model for text-to-video, first-frame and first+last-frame animation, loose multimodal references, audio-driven generation, and uploaded-video editing/extension. It renders 2-30s at fixed 30 fps with optional native audio, supports 480p/720p/1080p and 16:9/4:3/1:1/3:4/9:16, accepts up to 10 reference images, 5 reference videos, and 5 reference audios, and uses plain per-type prompt labels Image 1, Video 1, and Audio 1. Do not send negativePrompt. Use animate_photo for native first/last frames, generate_video for text or loose references, sound_to_video when audio is the primary driver, and video_to_video with controlMode=\"seedance-v2v\" for edits or extensions."
|
|
50
79
|
},
|
|
51
80
|
"generateAudio": {
|
|
52
81
|
"type": "boolean",
|
|
53
|
-
"description": "Whether
|
|
82
|
+
"description": "Whether the returned video should include generated/native audio. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. Supported by LTX, MiniMax H3, and Seedance; not supported by WAN or HappyHorse.\n\nWan 3 supports this toggle; omit it for audio-on by default or set false only for an explicitly silent result."
|
|
54
83
|
},
|
|
55
84
|
"referenceImageIndices": {
|
|
56
85
|
"type": "array",
|
|
57
86
|
"items": {
|
|
58
87
|
"type": "number"
|
|
59
88
|
},
|
|
60
|
-
"description": "
|
|
89
|
+
"description": "Seedance or MiniMax H3 R2V image references. Use negative indices for uploaded images and non-negative indices for generated image results. Seedance uses @Image tags. H3 uses <Picture 1>, <Picture 2>, and so on in selection order; these are loose references, not locked frames. H3 requires at least one visual across referenceImageIndices and referenceVideoIndices, so this array may be empty when a reference video is supplied; audio alone is invalid. Wan 3 loose images use Image 1, Image 2, and so on, with up to 10 images."
|
|
61
90
|
},
|
|
62
91
|
"referenceVideoIndices": {
|
|
63
92
|
"type": "array",
|
|
64
93
|
"items": {
|
|
65
94
|
"type": "number"
|
|
66
95
|
},
|
|
67
|
-
"description": "
|
|
96
|
+
"description": "Seedance or MiniMax H3 R2V loose video references. Use negative indices for uploaded videos and non-negative indices for generated video results. Seedance uses @Video tags; H3 uses <Video 1>, <Video 2>, and so on in selection order. A reference video can be the only visual input for H3 Ref2VA. Do not use this for source-video transforms; use video_to_video instead. Wan 3 loose videos use Video 1, Video 2, and so on, with up to 5 videos."
|
|
68
97
|
},
|
|
69
98
|
"referenceAudioIndices": {
|
|
70
99
|
"type": "array",
|
|
71
100
|
"items": {
|
|
72
101
|
"type": "number"
|
|
73
102
|
},
|
|
74
|
-
"description": "
|
|
103
|
+
"description": "Seedance or MiniMax H3 R2V loose audio references. Use negative indices for uploaded audio files and non-negative indices for generated audio results. Seedance uses @Audio tags; H3 uses <Audio 1>, <Audio 2>, and so on in selection order. H3 audio may accompany an image or video but cannot be the sole reference input. Wan 3 loose audios use Audio 1, Audio 2, and so on, with up to 5 audios."
|
|
75
104
|
},
|
|
76
105
|
"width": {
|
|
77
106
|
"type": "number",
|
|
78
|
-
"description": "Video width in pixels. LTX 2.
|
|
107
|
+
"description": "Video width in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Default resolution depends on model and quality tier: LTX Fast about 720p and High/Pro about 1080p; WAN Fast uses 480p short side and High/Pro uses 720p short side. Set width only when the user specifies an exact width or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target, not an instruction to make landscape 1280x720. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override the default media quality. Mappings when orientation is explicit: 480p landscape=854x480, 480p portrait=480x854, 720p landscape=1280x720, 720p portrait=720x1280, 1080p landscape=1920x1080, 1080p portrait=1080x1920, 4K landscape=3840x2160. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
79
108
|
},
|
|
80
109
|
"height": {
|
|
81
110
|
"type": "number",
|
|
82
|
-
"description": "Video height in pixels. LTX 2.
|
|
111
|
+
"description": "Video height in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Set height only when the user specifies an exact height or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target; do not convert it to landscape dimensions unless the user says landscape/horizontal/widescreen. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override Default Media Quality, including Pro. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
|
|
83
112
|
},
|
|
84
113
|
"targetResolution": {
|
|
85
114
|
"type": "number",
|
|
86
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-mini\" vs \"seedance2-5\"). Seedance 2.0 full supports 4K; Seedance Mini and Seedance 2.5 support 480p/720p only, so never set 1080p or 4K for \"seedance2-5\". Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
|
|
115
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-mini\" vs \"seedance2-5\"). Seedance 2.0 full supports 4K; Seedance Mini and Seedance 2.5 support 480p/720p only, so never set 1080p or 4K for \"seedance2-5\". Wan 3 supports exactly 480p, 720p, and 1080p. HappyHorse supports only 720p and 1080p. Never set 4K for Wan 3 or HappyHorse. MiniMax H3 renders inside a 1344x768 pixel budget on a 32px grid, so use 768 for H3 and never 1080p or 4K. Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, Wan 3, HappyHorse, or MiniMax H3, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
|
|
87
116
|
},
|
|
88
117
|
"numberOfVariations": {
|
|
89
118
|
"type": "number",
|
|
90
|
-
"description": "Number of variations (1-16). Use 1 unless user explicitly requests multiple separate video outputs. For Seedance, default to 1
|
|
119
|
+
"description": "Number of variations (1-16). Use with one Dynamic Prompt branch for multiple prompt-only takes that share the same references, model, duration, dimensions, and parameters. This creates one Sogni project with multiple jobs. Use 1 unless the user explicitly requests multiple separate video outputs. For Seedance, default to 1 unless the user explicitly requests separate outputs.",
|
|
91
120
|
"minimum": 1,
|
|
92
121
|
"maximum": 16
|
|
93
122
|
},
|
|
@@ -98,6 +127,25 @@
|
|
|
98
127
|
"voicePersonaName": {
|
|
99
128
|
"type": "string",
|
|
100
129
|
"description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one described in the video (e.g. \"David\" narrates a scene featuring Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
|
|
130
|
+
},
|
|
131
|
+
"loras": {
|
|
132
|
+
"type": "array",
|
|
133
|
+
"minItems": 1,
|
|
134
|
+
"maxItems": 8,
|
|
135
|
+
"items": {
|
|
136
|
+
"type": "string",
|
|
137
|
+
"minLength": 1
|
|
138
|
+
},
|
|
139
|
+
"description": "Ordered LoRA IDs to apply to a MiniMax H3 render. Use only when the user explicitly asks for a LoRA or for an effect one of these names describes. Stack up to 8 in one request; order matters because the adapters apply in sequence and do not commute. Keep this array positionally aligned with loraStrengths. The first render with an uncached LoRA takes longer to start while the worker downloads it.\n\nAccepted only when videoModel is one of \"minimax-h3-t2v\", \"minimax-h3-t2v-turbo\", \"minimax-h3-r2v\", \"minimax-h3-r2v-turbo\". Every other video model on this tool loads no LoRAs and silently ignores these arrays, so set videoModel to an H3 mode in the same call when the user asks for one.\n\nOne LoRA is published for MiniMax H3 today: h3-realism-people (fal), a realism pass trained on live-action footage of people. It restores skin texture and pores, stray hairs, fabric weave and a fine sensor grain that the base model smooths away, and holds up in close-up. It needs its trigger word: put r34l1sm near the FRONT of the prompt, or the render comes back as ordinary H3 with no error. Exact ranges and any LoRA published since: GET /v1/loras/comfy?modelId=<model>. Do not invent ids."
|
|
140
|
+
},
|
|
141
|
+
"loraStrengths": {
|
|
142
|
+
"type": "array",
|
|
143
|
+
"minItems": 1,
|
|
144
|
+
"maxItems": 8,
|
|
145
|
+
"items": {
|
|
146
|
+
"type": "number"
|
|
147
|
+
},
|
|
148
|
+
"description": "Strength for each LoRA in loras, in the same order. Omitting the array applies 1.0 to every LoRA, which is NOT the catalog default and for h3-realism-people is already at the top of its band, so send explicit values. Video LoRAs are positive-only — unlike the bipolar Krea 2 image sliders, a negative value is not an inverse effect and 0 is off. h3-realism-people takes 0-2 and its catalog default is 0.8; 0.6-1 is the usable band. It also pulls the camera in as it climbs: at 1.5 and above the shot reliably recomposes and the grade darkens, which on an image-conditioned mode can crop the subject out of the frame the user supplied. Raise it above 1 only when the user asks for more, and prefer the default when they supplied a first or last frame."
|
|
101
149
|
}
|
|
102
150
|
},
|
|
103
151
|
"required": [
|