@sogni-ai/sogni-protocol 1.0.0-alpha.24 → 1.0.0-alpha.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,26 +12,26 @@
12
12
  "type": "function",
13
13
  "function": {
14
14
  "name": "enhance_prompt",
15
- "description": "Enhance or adapt a source prompt into a model-ready image, video, music, or edit prompt. Use for prompt expansion and model-specific prompt preparation. Do not use for lyrics or full scripts.",
15
+ "description": "Author or adapt a directly runnable image, video, or image-edit prompt in an exact active model's native format. Use this when the requested text deliverable is explicitly a prompt for a named media model, including a commercial prompt; always pass destination_model and the matching target_output. Unknown, sunset, or omitted models fail closed. Do not use for lyrics, music composition, screenplays, storyboards, treatments, or ad scripts.",
16
16
  "parameters": {
17
17
  "type": "object",
18
18
  "additionalProperties": false,
19
- "required": ["prompt"],
19
+ "required": ["prompt", "target_output", "destination_model"],
20
20
  "properties": {
21
21
  "prompt": { "type": "string", "description": "The source prompt, rough idea, or prompt revision request to enhance." },
22
- "target_output": { "type": "string", "enum": ["image_prompt", "video_prompt", "music_prompt", "edit_prompt", "model_prompt", "general_prompt"], "description": "The kind of prompt artifact to produce." },
23
- "destination_model": { "type": "string", "description": "Optional destination model selector, such as seedance2, ltx23, wan22, gpt-image-2, or sdxl." },
24
- "destination_tool": { "type": "string", "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, video_to_video, or generate_music." },
25
- "prompting_type": { "type": "string", "enum": ["flux", "sdxl", "sd15", "pony", "fast", "sd3", "editing", "video"], "description": "Optional image-prompting family when producing an image prompt." },
26
- "model_title": { "type": "string", "description": "Optional human-readable target model name for image prompt guidance." },
22
+ "target_output": { "type": "string", "enum": ["image_prompt", "video_prompt", "edit_prompt", "model_prompt"], "description": "The model-specific prompt artifact to produce. Use model_prompt only when destination_model unambiguously identifies the modality." },
23
+ "destination_model": { "type": "string", "description": "Required exact active destination model selector, such as seedance2, ltx25, ltx23, wan22, minimax-h3-turbo, qwen, gpt-image-2, krea-2-turbo, chroma-v46-flash, or sdxl. Unknown and sunset models fail closed." },
24
+ "destination_tool": { "type": "string", "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, or video_to_video." },
25
+ "prompting_type": { "type": "string", "enum": ["flux", "sdxl", "sd15", "pony", "fast", "sd3", "editing", "video"], "description": "Deprecated compatibility metadata. It never selects or overrides the prompt grammar; destination_model is authoritative." },
26
+ "model_title": { "type": "string", "description": "Deprecated compatibility label. It never selects or overrides the prompt grammar; destination_model is authoritative." },
27
27
  "style_prompt": { "type": "string", "description": "Optional current style, brand, or prompt context to complement without repeating." },
28
28
  "prompt_mode": { "type": "string", "enum": ["auto", "preserve", "expand", "compress", "validate", "payload"], "description": "Optional model prompt adaptation mode." },
29
- "duration_seconds": { "type": "number", "minimum": 1, "maximum": 300, "description": "Requested runtime when enhancing a video or music prompt." },
29
+ "duration_seconds": { "type": "number", "minimum": 1, "maximum": 300, "description": "Requested runtime when authoring a video prompt." },
30
30
  "aspect_ratio": { "type": "string", "description": "Optional target aspect ratio, such as 16:9, 9:16, 1:1, 4:5, or 21:9." },
31
31
  "assets": {
32
32
  "type": "array",
33
- "maxItems": 12,
34
- "description": "Optional available assets the enhanced prompt may reference.",
33
+ "maxItems": 50,
34
+ "description": "Optional available assets the authored prompt may reference. The selected exact model contract applies its per-modality and total limits.",
35
35
  "items": {
36
36
  "type": "object",
37
37
  "additionalProperties": false,
@@ -89,7 +89,7 @@
89
89
  "type": "function",
90
90
  "function": {
91
91
  "name": "compose_script",
92
- "description": "Compose scripts, storyboards, video prompts, ad concepts, trailers, social shorts, campaign beats, and talking-head plans. Use for creative writing artifacts. Do not use for lyrics or simple prompt expansion.",
92
+ "description": "Compose scripts, screenplays, storyboards, treatments, ad concepts, trailers, social shorts, campaign beats, and talking-head plans. Use for creative-writing artifacts that are not directly runnable model prompts. When the user explicitly asks for a prompt for a named image or video model, use enhance_prompt instead, even if the prompt is for a commercial. Do not use for lyrics.",
93
93
  "parameters": {
94
94
  "type": "object",
95
95
  "additionalProperties": false,
@@ -178,7 +178,7 @@
178
178
  "minimax-h3-r2v",
179
179
  "minimax-h3-r2v-turbo"
180
180
  ],
181
- "description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0 and Mini accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with up to 50 reference media files total, subject to those per-modality caps. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
181
+ "description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0 and Mini accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with up to 50 reference media files total, subject to those per-modality caps. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
182
182
  },
183
183
  "generateAudio": {
184
184
  "type": "boolean",
@@ -573,7 +573,7 @@
573
573
  "minimax-h3-flf2v",
574
574
  "minimax-h3-flf2v-turbo"
575
575
  ],
576
- "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
576
+ "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
577
577
  },
578
578
  "generateAudio": {
579
579
  "type": "boolean",
@@ -990,7 +990,7 @@
990
990
  "ltx23-ia2v",
991
991
  "ltx23-a2v"
992
992
  ],
993
- "description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast/HQ use official distilled INT8 and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
993
+ "description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflows; Dev is withheld until upstream publishes and Sogni validates official ComfyUI Dev recipes. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
994
994
  },
995
995
  "generateAudio": {
996
996
  "type": "boolean",
@@ -158,7 +158,7 @@
158
158
  "minimax-h3-r2v",
159
159
  "minimax-h3-r2v-turbo"
160
160
  ],
161
- "description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0 and Mini accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with up to 50 reference media files total, subject to those per-modality caps. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
161
+ "description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0 and Mini accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with up to 50 reference media files total, subject to those per-modality caps. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
162
162
  },
163
163
  "generateAudio": {
164
164
  "type": "boolean",
@@ -553,7 +553,7 @@
553
553
  "minimax-h3-flf2v",
554
554
  "minimax-h3-flf2v-turbo"
555
555
  ],
556
- "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
556
+ "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
557
557
  },
558
558
  "generateAudio": {
559
559
  "type": "boolean",
@@ -970,7 +970,7 @@
970
970
  "ltx23-ia2v",
971
971
  "ltx23-a2v"
972
972
  ],
973
- "description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast/HQ use official distilled INT8 and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
973
+ "description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflows; Dev is withheld until upstream publishes and Sogni validates official ComfyUI Dev recipes. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
974
974
  },
975
975
  "generateAudio": {
976
976
  "type": "boolean",
@@ -1314,12 +1314,14 @@
1314
1314
  "type": "function",
1315
1315
  "function": {
1316
1316
  "name": "enhance_prompt",
1317
- "description": "Enhance or adapt a source prompt into a model-ready image, video, music, or edit prompt. Use for prompt expansion and model-specific prompt preparation. Do not use for lyrics or full scripts.",
1317
+ "description": "Author or adapt a directly runnable image, video, or image-edit prompt in an exact active model's native format. Use this when the requested text deliverable is explicitly a prompt for a named media model, including a commercial prompt; always pass destination_model and the matching target_output. Unknown, sunset, or omitted models fail closed. Do not use for lyrics, music composition, screenplays, storyboards, treatments, or ad scripts.",
1318
1318
  "parameters": {
1319
1319
  "type": "object",
1320
1320
  "additionalProperties": false,
1321
1321
  "required": [
1322
- "prompt"
1322
+ "prompt",
1323
+ "target_output",
1324
+ "destination_model"
1323
1325
  ],
1324
1326
  "properties": {
1325
1327
  "prompt": {
@@ -1331,20 +1333,18 @@
1331
1333
  "enum": [
1332
1334
  "image_prompt",
1333
1335
  "video_prompt",
1334
- "music_prompt",
1335
1336
  "edit_prompt",
1336
- "model_prompt",
1337
- "general_prompt"
1337
+ "model_prompt"
1338
1338
  ],
1339
- "description": "The kind of prompt artifact to produce."
1339
+ "description": "The model-specific prompt artifact to produce. Use model_prompt only when destination_model unambiguously identifies the modality."
1340
1340
  },
1341
1341
  "destination_model": {
1342
1342
  "type": "string",
1343
- "description": "Optional destination model selector, such as seedance2, ltx23, wan22, gpt-image-2, or sdxl."
1343
+ "description": "Required exact active destination model selector, such as seedance2, ltx25, ltx23, wan22, minimax-h3-turbo, qwen, gpt-image-2, krea-2-turbo, chroma-v46-flash, or sdxl. Unknown and sunset models fail closed."
1344
1344
  },
1345
1345
  "destination_tool": {
1346
1346
  "type": "string",
1347
- "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, video_to_video, or generate_music."
1347
+ "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, or video_to_video."
1348
1348
  },
1349
1349
  "prompting_type": {
1350
1350
  "type": "string",
@@ -1358,11 +1358,11 @@
1358
1358
  "editing",
1359
1359
  "video"
1360
1360
  ],
1361
- "description": "Optional image-prompting family when producing an image prompt."
1361
+ "description": "Deprecated compatibility metadata. It never selects or overrides the prompt grammar; destination_model is authoritative."
1362
1362
  },
1363
1363
  "model_title": {
1364
1364
  "type": "string",
1365
- "description": "Optional human-readable target model name for image prompt guidance."
1365
+ "description": "Deprecated compatibility label. It never selects or overrides the prompt grammar; destination_model is authoritative."
1366
1366
  },
1367
1367
  "style_prompt": {
1368
1368
  "type": "string",
@@ -1384,7 +1384,7 @@
1384
1384
  "type": "number",
1385
1385
  "minimum": 1,
1386
1386
  "maximum": 300,
1387
- "description": "Requested runtime when enhancing a video or music prompt."
1387
+ "description": "Requested runtime when authoring a video prompt."
1388
1388
  },
1389
1389
  "aspect_ratio": {
1390
1390
  "type": "string",
@@ -1392,8 +1392,8 @@
1392
1392
  },
1393
1393
  "assets": {
1394
1394
  "type": "array",
1395
- "maxItems": 12,
1396
- "description": "Optional available assets the enhanced prompt may reference.",
1395
+ "maxItems": 50,
1396
+ "description": "Optional available assets the authored prompt may reference. The selected exact model contract applies its per-modality and total limits.",
1397
1397
  "items": {
1398
1398
  "type": "object",
1399
1399
  "additionalProperties": false,
@@ -1510,7 +1510,7 @@
1510
1510
  "type": "function",
1511
1511
  "function": {
1512
1512
  "name": "compose_script",
1513
- "description": "Compose scripts, storyboards, video prompts, ad concepts, trailers, social shorts, campaign beats, and talking-head plans. Use for creative writing artifacts. Do not use for lyrics or simple prompt expansion.",
1513
+ "description": "Compose scripts, screenplays, storyboards, treatments, ad concepts, trailers, social shorts, campaign beats, and talking-head plans. Use for creative-writing artifacts that are not directly runnable model prompts. When the user explicitly asks for a prompt for a named image or video model, use enhance_prompt instead, even if the prompt is for a commercial. Do not use for lyrics.",
1514
1514
  "parameters": {
1515
1515
  "type": "object",
1516
1516
  "additionalProperties": false,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@sogni-ai/sogni-protocol",
3
- "version": "1.0.0-alpha.24",
3
+ "version": "1.0.0-alpha.26",
4
4
  "description": "Language-neutral protocol artifacts for the Sogni ecosystem: tool schemas, prompts, OpenAI tool manifests, and enums. Consumed by every Sogni SDK (TypeScript, Swift, and future Python/Kotlin/Rust SDKs) so contracts stay in lockstep across languages.",
5
5
  "keywords": [
6
6
  "sogni",
@@ -1,8 +1,8 @@
1
1
  {
2
2
  "contractId": "compose_script_v1",
3
- "version": "1.0.0",
3
+ "version": "1.1.0",
4
4
  "toolName": "compose_script",
5
- "baseDescription": "compose_script composes scripts, storyboards, video prompts, ad concepts, trailers, social\nshorts, campaign beats, and talking-head plans. Use for creative writing artifacts where\nthe user wants prose, scenes, beats, or a script-shaped deliverable.\n\nDo not use compose_script for song lyrics — use compose_lyrics. Do not use it for simple\nprompt expansion — use enhance_prompt. Do not use it to plan a runnable multi-step\nworkflow — use compose_workflow or compose_workflow_template.\n\ncompose_script returns a creative-writing deliverable. When the script is meant to feed a\ndownstream video tool, pair it with destination_tool/destination_model so the output is\ntuned for that pipeline; the caller still has to invoke the generation tool itself.",
5
+ "baseDescription": "compose_script composes scripts, screenplays, storyboards, treatments, ad concepts, trailers,\nsocial shorts, campaign beats, and talking-head plans. Use it for prose, scenes, beats, or\nother script-shaped creative-writing deliverables.\n\nWhen the user explicitly asks for a directly runnable prompt for a named image or video\nmodel, use enhance_prompt instead, even when the subject is a commercial. Do not use\ncompose_script for song lyrics, instrumental composition, simple prompt authoring, or a\nrunnable multi-step workflow.\n\ncompose_script returns a creative-writing deliverable. When that deliverable will later feed\na video pipeline, destination_tool and destination_model may record the intended handoff;\nthey do not turn the result into the model's native prompt contract.",
6
6
  "parameterDocs": {
7
7
  "brief": "The creative writing brief, story idea, product concept, video idea, or revision request.",
8
8
  "script_type": "The kind of script or creative writing artifact to produce. One of video_prompt, screenplay, storyboard, ad_script, trailer, social_short, talking_head, campaign, or revision.",
@@ -1,20 +1,20 @@
1
1
  {
2
2
  "contractId": "enhance_prompt_v1",
3
- "version": "1.0.0",
3
+ "version": "1.1.0",
4
4
  "toolName": "enhance_prompt",
5
- "baseDescription": "enhance_prompt enhances or adapts a source prompt into a model-ready image, video, music, or\nedit prompt. Use for prompt expansion and model-specific prompt preparation when the caller\nhas a rough idea, a revision request, or a prompt that needs to be tuned for a specific\ndownstream model or generation tool.\n\nDo not use enhance_prompt for full creative-writing artifacts. Use compose_script for\nscripts, storyboards, ad concepts, trailers, and talking-head plans. Use compose_lyrics for\nsong lyrics and compose_instrumental for instrumental music structure. Do not use\nenhance_prompt to plan a multi-step workflow — use compose_workflow or\ncompose_workflow_template instead.\n\nenhance_prompt returns prompt text. The caller is responsible for handing the enhanced\nprompt to the chosen downstream generation tool (generate_image, edit_image,\ngenerate_video, animate_photo, sound_to_video, video_to_video, generate_music, etc.).",
5
+ "baseDescription": "enhance_prompt authors a directly runnable image, video, or image-edit prompt in the native\nformat of one exact active destination model. Use it when the requested text deliverable is\nexplicitly a prompt for a named media model, including a commercial prompt. The caller must\nprovide destination_model and target_output; unknown, sunset, and omitted models fail closed.\n\nDo not use enhance_prompt for model-neutral prompts, music composition, lyrics, screenplays,\nstoryboards, treatments, ad scripts, or multi-step workflow plans. Use compose_script for\ncreative-writing artifacts, compose_lyrics or compose_instrumental for music, and\ncompose_workflow or compose_workflow_template for workflows.\n\nenhance_prompt returns prompt text only. The caller is responsible for handing it to the\nmatching downstream generation tool.",
6
6
  "parameterDocs": {
7
7
  "prompt": "The source prompt, rough idea, or prompt revision request to enhance.",
8
- "target_output": "The kind of prompt artifact to produce. One of image_prompt, video_prompt, music_prompt, edit_prompt, model_prompt, or general_prompt.",
9
- "destination_model": "Optional destination model selector, such as seedance2, ltx25, ltx23, wan22, gpt-image-2, or sdxl.",
10
- "destination_tool": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, video_to_video, or generate_music.",
11
- "prompting_type": "Optional image-prompting family when producing an image prompt. One of flux, sdxl, sd15, pony, fast, sd3, editing, or video.",
12
- "model_title": "Optional human-readable target model name for image prompt guidance.",
8
+ "target_output": "Required model-specific artifact kind: image_prompt, video_prompt, edit_prompt, or model_prompt. Use model_prompt only when destination_model identifies the modality.",
9
+ "destination_model": "Required exact active model selector. Unknown and sunset models fail closed.",
10
+ "destination_tool": "Optional matching downstream generation tool for the authored prompt.",
11
+ "prompting_type": "Deprecated compatibility metadata. It never selects or overrides the prompt grammar; destination_model is authoritative.",
12
+ "model_title": "Deprecated compatibility label. It never selects or overrides the prompt grammar; destination_model is authoritative.",
13
13
  "style_prompt": "Optional current style, brand, or prompt context to complement without repeating.",
14
14
  "prompt_mode": "Optional model prompt adaptation mode. One of auto, preserve, expand, compress, validate, or payload.",
15
- "duration_seconds": "Requested runtime in seconds (1-300) when enhancing a video or music prompt.",
15
+ "duration_seconds": "Requested runtime in seconds (1-300) when authoring a video prompt.",
16
16
  "aspect_ratio": "Optional target aspect ratio, such as 16:9, 9:16, 1:1, 4:5, or 21:9.",
17
- "assets": "Optional available assets (max 12) the enhanced prompt may reference. Each entry needs media_type (image/video/audio) and may include id, label, role, and url.",
17
+ "assets": "Optional available assets (schema max 50) the prompt may reference. The exact destination model applies smaller per-modality and total limits.",
18
18
  "constraints": "Optional production, brand, model, or user constraints to preserve."
19
19
  }
20
20
  }
@@ -32,7 +32,7 @@
32
32
  "minimax-h3-flf2v",
33
33
  "minimax-h3-flf2v-turbo"
34
34
  ],
35
- "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
35
+ "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
36
36
  },
37
37
  "generateAudio": {
38
38
  "type": "boolean",
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "title": "compose_script tool schema",
3
3
  "schemaVersion": "2026-05-14.1",
4
- "description": "Synchronous creative writing utility for scripts, storyboards, video prompts, ad concepts, trailers, social shorts, campaign beats, and talking-head plans.",
4
+ "description": "Synchronous creative-writing utility for scripts, screenplays, storyboards, treatments, ad concepts, trailers, social shorts, campaign beats, and talking-head plans. Directly runnable prompts for named media models belong to enhance_prompt.",
5
5
  "type": "object",
6
6
  "additionalProperties": false,
7
7
  "required": ["brief"],
@@ -1,10 +1,10 @@
1
1
  {
2
2
  "title": "enhance_prompt tool schema",
3
3
  "schemaVersion": "2026-05-14.1",
4
- "description": "Synchronous prompt enhancement utility for expanding or adapting a source prompt into a model-ready image, video, music, or edit prompt.",
4
+ "description": "Synchronous prompt-authoring utility for expanding or adapting a source prompt into an exact active image, video, or image-edit model's native prompt format.",
5
5
  "type": "object",
6
6
  "additionalProperties": false,
7
- "required": ["prompt"],
7
+ "required": ["prompt", "target_output", "destination_model"],
8
8
  "properties": {
9
9
  "prompt": {
10
10
  "type": "string",
@@ -12,25 +12,25 @@
12
12
  },
13
13
  "target_output": {
14
14
  "type": "string",
15
- "enum": ["image_prompt", "video_prompt", "music_prompt", "edit_prompt", "model_prompt", "general_prompt"],
16
- "description": "The kind of prompt artifact to produce."
15
+ "enum": ["image_prompt", "video_prompt", "edit_prompt", "model_prompt"],
16
+ "description": "The model-specific prompt artifact to produce. Use model_prompt only when destination_model unambiguously identifies the modality."
17
17
  },
18
18
  "destination_model": {
19
19
  "type": "string",
20
- "description": "Optional destination model selector, such as seedance2, ltx25, ltx23, wan22, gpt-image-2, or sdxl."
20
+ "description": "Required exact active destination model selector, such as seedance2, ltx25, ltx23, wan22, minimax-h3-turbo, qwen, gpt-image-2, krea-2-turbo, chroma-v46-flash, or sdxl. Unknown and sunset models fail closed."
21
21
  },
22
22
  "destination_tool": {
23
23
  "type": "string",
24
- "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, video_to_video, or generate_music."
24
+ "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, or video_to_video."
25
25
  },
26
26
  "prompting_type": {
27
27
  "type": "string",
28
28
  "enum": ["flux", "sdxl", "sd15", "pony", "fast", "sd3", "editing", "video"],
29
- "description": "Optional image-prompting family when producing an image prompt."
29
+ "description": "Deprecated compatibility metadata. It never selects or overrides the prompt grammar; destination_model is authoritative."
30
30
  },
31
31
  "model_title": {
32
32
  "type": "string",
33
- "description": "Optional human-readable target model name for image prompt guidance."
33
+ "description": "Deprecated compatibility label. It never selects or overrides the prompt grammar; destination_model is authoritative."
34
34
  },
35
35
  "style_prompt": {
36
36
  "type": "string",
@@ -45,7 +45,7 @@
45
45
  "type": "number",
46
46
  "minimum": 1,
47
47
  "maximum": 300,
48
- "description": "Requested runtime when enhancing a video or music prompt."
48
+ "description": "Requested runtime when authoring a video prompt."
49
49
  },
50
50
  "aspect_ratio": {
51
51
  "type": "string",
@@ -53,8 +53,8 @@
53
53
  },
54
54
  "assets": {
55
55
  "type": "array",
56
- "maxItems": 12,
57
- "description": "Optional available assets the enhanced prompt may reference.",
56
+ "maxItems": 50,
57
+ "description": "Optional available assets the authored prompt may reference. The selected exact model contract applies its per-modality and total limits.",
58
58
  "items": {
59
59
  "type": "object",
60
60
  "additionalProperties": false,
@@ -46,7 +46,7 @@
46
46
  "minimax-h3-r2v",
47
47
  "minimax-h3-r2v-turbo"
48
48
  ],
49
- "description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0 and Mini accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with up to 50 reference media files total, subject to those per-modality caps. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
49
+ "description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow; Dev is withheld until upstream publishes and Sogni validates an official ComfyUI Dev recipe. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0 and Mini accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with up to 50 reference media files total, subject to those per-modality caps. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
50
50
  },
51
51
  "generateAudio": {
52
52
  "type": "boolean",
@@ -50,7 +50,7 @@
50
50
  "ltx23-ia2v",
51
51
  "ltx23-a2v"
52
52
  ],
53
- "description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast/HQ use official distilled INT8 and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
53
+ "description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflows; Dev is withheld until upstream publishes and Sogni validates official ComfyUI Dev recipes. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
54
54
  },
55
55
  "generateAudio": {
56
56
  "type": "boolean",
@@ -3,13 +3,13 @@
3
3
  "$id": "https://schemas.sogni.ai/creative-agent/2026-04-27.1/tools/video_to_video.schema.json",
4
4
  "title": "video_to_video arguments",
5
5
  "schemaVersion": "2026-04-27.1",
6
- "description": "Transform an existing video using WAN 2.2 Animate, LTX 2.5 V2V controls by default, LTX 2.3 as rollback, or Seedance V2V when explicitly requested. LTX 2.5 distilled supports canny/pose/depth/detailer/inpaint/outpaint; Dev + Speed LoRA supports canny/pose/depth/detailer. Requires an uploaded video.",
6
+ "description": "Transform an existing video using WAN 2.2 Animate, LTX 2.5 V2V controls by default, LTX 2.3 as rollback, or Seedance V2V when explicitly requested. LTX 2.5 Fast, HQ, and Pro use the release-validated official Distilled workflow for canny/pose/depth/detailer/inpaint/outpaint; Dev is not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Requires an uploaded video.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "properties": {
10
10
  "prompt": {
11
11
  "type": "string",
12
- "description": "Describe the TARGET appearance, motion, dialogue, audio, and style in positive present-tense language. For LTX 2.5 (default) or LTX 2.3 rollback canny/depth/pose modes, the source preserves the selected structure or motion, so emphasize style, atmosphere, lighting, texture, color, scale, and pacing. Canny preserves edges; pose preserves skeletal motion; depth preserves 3D layout; detailer should describe the original content with quality qualifiers only. Distilled LTX 2.5 also supports inpaint and outpaint; Dev + Speed LoRA does not. For inpaint, describe only the regenerated region. For outpaint, describe the newly revealed area consistently with the source. For Seedance V2V, use natural prose and describe the target transformation holistically."
12
+ "description": "Describe the TARGET appearance, motion, dialogue, audio, and style in positive present-tense language. For LTX 2.5 (default) or LTX 2.3 rollback canny/depth/pose modes, the source preserves the selected structure or motion, so emphasize style, atmosphere, lighting, texture, color, scale, and pacing. Canny preserves edges; pose preserves skeletal motion; depth preserves 3D layout; detailer should describe the original content with quality qualifiers only. The validated Distilled LTX 2.5 path supports inpaint and outpaint; Dev is not publicly routed. For inpaint, describe only the regenerated region. For outpaint, describe the newly revealed area consistently with the source. For Seedance V2V, use natural prose and describe the target transformation holistically."
13
13
  },
14
14
  "expandPrompt": {
15
15
  "type": "boolean",
package/version.json CHANGED
@@ -1,4 +1,4 @@
1
1
  {
2
- "protocolVersion": "2.1.0",
2
+ "protocolVersion": "3.0.0",
3
3
  "description": "Sogni protocol artifact version. SDKs may refuse to operate against a protocolVersion they were not built for. Bump the major when removing or renaming any schema / enum / manifest field; bump the minor when adding new optional fields or new tools; bump the patch for description / prose changes only."
4
4
  }