@sogni-ai/sogni-protocol 1.0.0-alpha.2 → 1.0.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +10 -1
  2. package/catalogs/audio-models.json +68 -7
  3. package/catalogs/quality-presets.json +3 -3
  4. package/catalogs/seedance-reference-limits.json +35 -0
  5. package/enums/tool-names.json +2 -0
  6. package/manifests/composition-tools.json +3 -3
  7. package/manifests/generation-tools.json +153 -77
  8. package/manifests/openai-tools.json +140 -65
  9. package/package.json +1 -1
  10. package/prompts/tools/animate_photo.json +1 -1
  11. package/prompts/tools/compose_script.json +1 -1
  12. package/prompts/tools/compose_workflow.json +1 -1
  13. package/prompts/tools/compose_workflow_template.json +1 -1
  14. package/prompts/tools/edit_image.json +1 -1
  15. package/prompts/tools/enhance_prompt.json +1 -1
  16. package/prompts/tools/extend_video.json +2 -2
  17. package/prompts/tools/generate_image.json +2 -2
  18. package/prompts/tools/generate_video.json +1 -1
  19. package/prompts/tools/map_assets_for_model.json +1 -1
  20. package/prompts/tools/replace_video_segment.json +2 -2
  21. package/prompts/tools/resolve_personas.json +1 -1
  22. package/prompts/tools/sound_to_video.json +3 -2
  23. package/prompts/tools/video_to_video.json +2 -2
  24. package/schemas/agent/intent-input.schema.json +128 -0
  25. package/schemas/agent/turn-analysis.schema.json +75 -0
  26. package/schemas/artifacts/artifact-graph.schema.json +42 -0
  27. package/schemas/artifacts/artifact-node.schema.json +137 -0
  28. package/schemas/billing/spend-gate.schema.json +151 -0
  29. package/schemas/billing/workflow-authorization.schema.json +83 -0
  30. package/schemas/events/run-event.schema.json +122 -0
  31. package/schemas/tools/animate_photo.schema.json +23 -12
  32. package/schemas/tools/compose_script.schema.json +1 -1
  33. package/schemas/tools/compose_workflow.schema.json +2 -2
  34. package/schemas/tools/compose_workflow_template.schema.json +2 -2
  35. package/schemas/tools/edit_image.schema.json +8 -7
  36. package/schemas/tools/enhance_prompt.schema.json +1 -1
  37. package/schemas/tools/extend_video.schema.json +8 -5
  38. package/schemas/tools/generate_image.schema.json +12 -9
  39. package/schemas/tools/generate_music.schema.json +3 -2
  40. package/schemas/tools/generate_video.schema.json +24 -14
  41. package/schemas/tools/replace_video_segment.schema.json +5 -2
  42. package/schemas/tools/sound_to_video.schema.json +16 -8
  43. package/schemas/tools/tool-metadata.schema.json +78 -0
  44. package/schemas/tools/upscale_image.schema.json +31 -0
  45. package/schemas/tools/video_to_video.schema.json +13 -10
  46. package/schemas/workflows/durable-workflow-run.schema.json +1 -0
  47. package/version.json +1 -1
@@ -3,13 +3,13 @@
3
3
  "$id": "https://schemas.sogni.ai/creative-agent/2026-04-27.1/tools/animate_photo.schema.json",
4
4
  "title": "animate_photo arguments",
5
5
  "schemaVersion": "2026-04-27.1",
6
- "description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.3 or WAN 2.2. Do NOT use this tool for seedance2 or seedance2-fast. Seedance 2.0 media references must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.3 generates audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 — always use ltx23 when persona voice is requested (WAN 2.2 does not support voice identity). PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, FIRST generate both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, THEN call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. Never generate the two frames in separate tool calls. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Only call once you have clear creative intent. N-VIDEOS PATTERN — ALWAYS BATCH IN ONE CALL: When the user wants N video versions or a multi-segment stitched non-Seedance video, NEVER call animate_photo N times. Always use sourceImageIndices in a single call so all N projects run in parallel. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips (30s total = 2 clips × 15s), not 5×6s or 6×5s. Two flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image instead of generated source images, use sourceImageIndices=[-1,-1,...] with one -1 per segment. For a long or multi-segment video from a single supplied/uploaded image WITHOUT a requested image/keyframe/version generation stage, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when each clip has DIFFERENT dialogue, jokes, narration, or motion (e.g. \"4 videos where each tells a different joke\"), pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. The fan-out launches all N projects in parallel with their respective per-clip prompts. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame instead.",
6
+ "description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.5 by default, LTX 2.3 as rollback, or WAN 2.2. Do NOT use this tool for seedance2, seedance2-mini, seedance2-fast, or seedance2-5. Seedance media references — including Seedance 2.5 first-and-last-frame requests — must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.5 and LTX 2.3 generate audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Avoid placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, prefer generating both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, then call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. If the user explicitly wants separately created frame assets, preserve that staged instruction while keeping indices correct. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Call once you have clear creative intent. N-VIDEOS PATTERN: Avoid sequential animate_photo calls for N outputs. For a single fixed source/end frame where only prompt text varies, use sourceImageIndex + numberOfVariations=N + one Dynamic Prompt branch in prompt so Sogni submits one project with multiple jobs. If the user explicitly asks for Dynamic Prompt or Dynamic Template syntax, prefer this one-project path whenever every output uses the same source/end frames and shared settings, even if they also ask to stitch the completed clips afterward. Use sourceImageIndices/prompts for multi-segment stitched non-Seedance video, different source/end assets, different audio windows, different durations/dimensions, isolated retry lifecycle, or other per-output parameters. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips on ltx23 (30s total = 2 clips × 15s) and 10-second clips on wan22 (60s total on wan22 = 6 clips × 10s; do NOT pick 4 clips × 15s on wan22 — the wan22 worker rejects clips longer than 10s). Multi-source flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image and only prompt text varies, use sourceImageIndex=-1, frameRole=\"both\" if requested, endImageIndex=-1 if requested, numberOfVariations=N, and one Dynamic Prompt branch in prompt. For a long scripted/dialogue/storyboard video from a single supplied/uploaded image where each segment needs isolated exact dialogue or per-segment wiring, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when source/end asset wiring or other per-output parameters differ, pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "properties": {
10
10
  "prompt": {
11
11
  "type": "string",
12
- "description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.3 generates audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
12
+ "description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.5 and LTX 2.3 generate audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same source/end frames and generation parameters and only prompt text varies. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
13
13
  },
14
14
  "expandPrompt": {
15
15
  "type": "boolean",
@@ -22,22 +22,33 @@
22
22
  "videoModel": {
23
23
  "type": "string",
24
24
  "enum": [
25
+ "ltx25",
25
26
  "ltx23",
26
- "wan22"
27
+ "wan22",
28
+ "happyhorse-1.1-i2v",
29
+ "happyhorse-1.1-r2v",
30
+ "minimax-h3-i2v",
31
+ "minimax-h3-i2v-turbo",
32
+ "minimax-h3-flf2v",
33
+ "minimax-h3-flf2v-turbo"
27
34
  ],
28
- "description": "Which video model to use. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"wan22\": Fast 4-step, simple motion, no audio. Use ltx23 for most requests. Use wan22 for quick simple motions without audio. Default: \"ltx23\". Do not set seedance2 or seedance2-fast here; use generate_video with referenceImageIndices and @Image role text for Seedance."
35
+ "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, seedance2-fast, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
36
+ },
37
+ "generateAudio": {
38
+ "type": "boolean",
39
+ "description": "Whether to include generated/native audio for audio-capable models. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. When false, the returned video has no audio track. Ignored by audio-less WAN."
29
40
  },
30
41
  "negativePrompt": {
31
42
  "type": "string",
32
- "description": "Optional negative prompt for LTX/Wan image-to-video models. Use only when the user explicitly states what should be avoided; do not use this field for Seedance workflows."
43
+ "description": "Advanced LTX 2.5/LTX 2.3/WAN only. All standard LTX 2.5 image and first/last-frame workflows accept this separate negative prompt. Use this field only when the user explicitly asks to set one. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt."
33
44
  },
34
45
  "duration": {
35
46
  "type": "number",
36
- "description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Range: 2-20."
47
+ "description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Per-model maximum: ltx25 and ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip."
37
48
  },
38
49
  "targetResolution": {
39
50
  "type": "number",
40
- "description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"1080p\" without exact pixels or an output orientation. This preserves the source image aspect ratio. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\" or \"720p landscape\", use exact-pixel aspectRatio instead."
51
+ "description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
41
52
  },
42
53
  "sourceImageIndex": {
43
54
  "type": "number",
@@ -50,7 +61,7 @@
50
61
  },
51
62
  "minItems": 1,
52
63
  "maxItems": 16,
53
- "description": "Array of source frame indices — one video is generated per entry as its own SDK project, all running in PARALLEL. Use 0-based non-negative result indices for generated images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Repeating -1 is allowed and is REQUIRED for multi-segment videos that reuse the same uploaded image as every segment's start frame. By default all projects share the `prompt`/`voice`/`duration`, but you can pass `prompts` (array) to give each clip its own dialogue/motion. ALWAYS use this for non-Seedance \"N videos\" or multi-segment request — never call animate_photo N times sequentially. Do NOT combine with `numberOfVariations`, `sourceImageIndex`, or frameRole=\"end\". You MAY combine with frameRole=\"both\" when clips need end frames. For adjacent transition chains across generated images, use sourceImageIndices=[start..end-1] and endImageIndices=[start+1..end] so N images produce N-1 transition clips. If the uploaded/original image starts the chain and generated results are the remaining frames, use sourceImageIndices=[-1,start..end-1] and endImageIndices=[start..end]. If the user supplies multiple uploaded images as the actual keyframe sequence, use adjacent negative uploaded indices, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4], endImageIndices=[-2,-3,-4,-5], frameRole=\"both\", prompts length 4, then stitch_video. If the user specifies transition motion, camera behavior, actions, dialogue, or audio, copy those instructions into every corresponding per-clip prompt; only invent a generic smooth transition when the user does not specify one. If the user asks for a seamless loop or final transition from the last image back to the first, close the chain by including the last image as a source and the first image as the final end frame, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4,-5], endImageIndices=[-2,-3,-4,-5,-1]. For generated scene keyframes that should each loop to themselves, omit endImageIndex/endImageIndices so each source image is also its own end frame. Set endImageIndex=-1 only when every sourceImageIndices entry is also -1 and every segment reuses the first uploaded image. Range: 1–16 indices. For generated image batches, values MUST be read from the latest edit_image/generate_image tool result's `startIndex` field. If startIndex=3 and 4 images were generated in that batch, pass `[3,4,5,6]` (NOT `[0,1,2,3]`). Do NOT assume generated indices start at 0 — they don't if there are prior results in the conversation."
64
+ "description": "Array of source frame indices — one video is generated per entry as its own SDK project, all running in PARALLEL. Use this when outcomes need different source images, different end frames, isolated retry lifecycle, or other per-clip asset wiring/parameters. If every outcome uses the same source/end frames and only prompt text differs, prefer sourceImageIndex with numberOfVariations=N and one Dynamic Prompt branch in `prompt` so Sogni creates one project with multiple jobs. Use 0-based non-negative result indices for generated images. Use negative indices for uploaded images: -1 = first/primary upload, -2 = second upload, -3 = third upload, etc. Repeating -1 is allowed for true multi-project workflows that intentionally reuse the same uploaded image while varying per-clip assets or parameters. By default all projects share the `prompt`/`voice`/`duration`, but you can pass `prompts` (array) to give each clip its own dialogue/motion when multi-project fan-out is required. Avoid sequential animate_photo calls for N outputs. Do NOT combine with `numberOfVariations` or `sourceImageIndex`. Use frameRole=\"end\" with sourceImageIndices only when the user explicitly says the repeated uploaded/generated image is the last/end frame for each clip and no first/start frame should be supplied; in that case omit endImageIndex/endImageIndices because each sourceImageIndices entry is the end frame. You MAY combine with frameRole=\"both\" when clips need start and end frames. For adjacent transition chains across generated images, use sourceImageIndices=[start..end-1] and endImageIndices=[start+1..end] so N images produce N-1 transition clips. If the uploaded/original image starts the chain and generated results are the remaining frames, use sourceImageIndices=[-1,start..end-1] and endImageIndices=[start..end]. If the user supplies multiple uploaded images as the actual keyframe sequence, use adjacent negative uploaded indices, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4], endImageIndices=[-2,-3,-4,-5], frameRole=\"both\", prompts length 4, then stitch_video. If the user specifies transition motion, camera behavior, actions, dialogue, or audio, copy those instructions into every corresponding per-clip prompt; only invent a generic smooth transition when the user does not specify one. If the user asks for a seamless loop or final transition from the last image back to the first, close the chain by including the last image as a source and the first image as the final end frame, e.g. 5 uploaded images become sourceImageIndices=[-1,-2,-3,-4,-5], endImageIndices=[-2,-3,-4,-5,-1]. For generated scene keyframes that should each loop to themselves, omit endImageIndex/endImageIndices so each source image is also its own end frame. Set endImageIndex=-1 only when every sourceImageIndices entry is also -1 and every segment reuses the first uploaded image. Range: 1–16 indices. For generated image batches, values MUST be read from the latest edit_image/generate_image tool result's `startIndex` field. If startIndex=3 and 4 images were generated in that batch, pass `[3,4,5,6]` (NOT `[0,1,2,3]`). Do NOT assume generated indices start at 0 — they don't if there are prior results in the conversation."
54
65
  },
55
66
  "prompts": {
56
67
  "type": "array",
@@ -59,11 +70,11 @@
59
70
  },
60
71
  "minItems": 1,
61
72
  "maxItems": 16,
62
- "description": "Per-clip prompts for fan-out — use when the user wants DIFFERENT dialogue, jokes, narration, or motion in each video. MUST be paired with `sourceImageIndices` and have the SAME length. Each entry is the full prompt for the corresponding source image. If a clip has speech, include exact spoken words in double quotes with stable speaker tags; do NOT write placeholders like \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". If you just wrote a script/table/storyboard, copy that clip's exact dialogue into this prompt. When named speakers appear in a multi-person reference image or generated keyframe, start each entry with one compact cast map that binds names to visible anchors before dialogue, e.g. Cast map: SPEAKER_A is the left person holding the prop; SPEAKER_B is the center person with the tablet; SPEAKER_C is the right person near the table. Then move directly into action/dialogue; do not describe those same people again as generic man/boy/girl/woman/character subjects. This prevents speaker tags from being assigned to the wrong visible character. When set, the top-level `prompt` parameter is ignored (still required by the schema — just pass any descriptive string, e.g. a brief summary of the batch). Example: 4 source images of a couple, \"make each video have a different joke\" → sourceImageIndices=[0,1,2,3], prompts=[\"Cast map: She is the left woman in the blue dress; He is the right man in the gray jacket. She says: \\\"Why did the scarecrow win an award?\\\" He grins.\", \"Cast map: He is the right man in the gray jacket; She is the left woman in the blue dress. He says: \\\"Because he was outstanding in his field!\\\" She laughs.\", \"...\", \"...\"]. Omit this when all clips share the same dialogue/motion (visual variety only) — fan-out will use the shared `prompt` for every clip."
73
+ "description": "Per-clip prompts for multi-project fan-out — use when each output needs different source/end assets, isolated retry lifecycle, or other per-clip wiring/parameters. If all outputs share the same source/end frames and only prompt text differs, put the full per-output prompts in ONE Dynamic Prompt branch in `prompt` and set numberOfVariations=N instead. When this field is required, it MUST be paired with `sourceImageIndices` and have the SAME length. Each entry is the full prompt for the corresponding source image. If a clip has speech, include exact spoken words in double quotes with stable speaker tags; do NOT write placeholders like \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". If you just wrote a script/table/storyboard, copy that clip's exact dialogue into this prompt. When named speakers appear in a multi-person reference image or generated keyframe, start each entry with one compact cast map that binds names to visible anchors before dialogue, e.g. Cast map: SPEAKER_A is the left person holding the prop; SPEAKER_B is the center person with the tablet; SPEAKER_C is the right person near the table. Then move directly into action/dialogue; do not describe those same people again as generic man/boy/girl/woman/character subjects. This prevents speaker tags from being assigned to the wrong visible character. When set, the top-level `prompt` parameter is ignored (still required by the schema — just pass any descriptive string, e.g. a brief summary of the batch). Example: 4 source images of a couple, \"make each video have a different joke\" → sourceImageIndices=[0,1,2,3], prompts=[\"Cast map: She is the left woman in the blue dress; He is the right man in the gray jacket. She says: \\\"Why did the scarecrow win an award?\\\" He grins.\", \"Cast map: He is the right man in the gray jacket; She is the left woman in the blue dress. He says: \\\"Because he was outstanding in his field!\\\" She laughs.\", \"...\", \"...\"]. Omit this whenever the same source/end assets and parameters can be represented as one Dynamic Prompt batch."
63
74
  },
64
75
  "numberOfVariations": {
65
76
  "type": "number",
66
- "description": "Number of variations (1-16). Use 1 unless user explicitly requests multiple separate video outputs.",
77
+ "description": "Number of variations (1-16). Use this with one Dynamic Prompt branch when the user explicitly requests multiple prompt-only takes from the same source/end frames. This creates one Sogni project with multiple jobs. Use 1 unless the user explicitly requests multiple separate video outputs; use sourceImageIndices/prompts instead only when assets or parameters differ per output.",
67
78
  "minimum": 1,
68
79
  "maximum": 16
69
80
  },
@@ -78,7 +89,7 @@
78
89
  "end",
79
90
  "both"
80
91
  ],
81
- "description": "How to use the source image(s) for non-Seedance video generation. \"start\" (default): image is the first frame — video animates forward from it. \"end\": image is the last frame — video leads up to it. \"both\": two images provided — interpolates between start and end frames. For single clips using \"both\", set sourceImageIndex to the start frame and endImageIndex to the end frame. For sourceImageIndices fan-out using repeated -1, use frameRole=\"both\" and set endImageIndex=-1 when every segment must use the uploaded image as the shared last frame. For adjacent generated-image transitions, use frameRole=\"both\" with matching sourceImageIndices and endImageIndices arrays. Omit endImageIndex/endImageIndices only when each source image should also be its own end frame. The handler inspects different start/end frames and generates a detailed transition prompt automatically."
92
+ "description": "How to use the source image(s) for non-Seedance video generation. \"start\" (default): image is the first frame — video animates forward from it. \"end\": image is the last frame — video leads up to it. \"both\": two images provided — interpolates between start and end frames. For single clips using \"both\", set sourceImageIndex to the start frame and endImageIndex to the end frame. For sourceImageIndices fan-out using repeated -1, use frameRole=\"end\" only when every clip should use that image as the last/end frame and no first/start frame should be supplied. Use frameRole=\"both\" and set endImageIndex=-1 when every segment must use the uploaded image as both its first and last frame. For adjacent generated-image transitions, use frameRole=\"both\" with matching sourceImageIndices and endImageIndices arrays. Omit endImageIndex/endImageIndices only when each source image should also be its own end frame. The handler inspects different start/end frames and generates a detailed transition prompt automatically."
82
93
  },
83
94
  "endImageIndex": {
84
95
  "type": "number",
@@ -95,7 +106,7 @@
95
106
  },
96
107
  "voicePersonaName": {
97
108
  "type": "string",
98
- "description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23."
109
+ "description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
99
110
  }
100
111
  },
101
112
  "required": [
@@ -17,7 +17,7 @@
17
17
  },
18
18
  "destination_model": {
19
19
  "type": "string",
20
- "description": "Optional destination video model selector, such as ltx23, wan22, or seedance2."
20
+ "description": "Optional destination video model selector, such as ltx25, ltx23, wan22, or seedance2."
21
21
  },
22
22
  "destination_tool": {
23
23
  "type": "string",
@@ -37,11 +37,11 @@
37
37
  "properties": {
38
38
  "image": {
39
39
  "type": "string",
40
- "description": "Preferred image model (e.g., 'flux2', 'gpt-image-2')."
40
+ "description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')."
41
41
  },
42
42
  "video": {
43
43
  "type": "string",
44
- "description": "Preferred video model (e.g., 'ltx23', 'wan22', 'seedance2')."
44
+ "description": "Preferred video model (e.g., 'ltx25', 'ltx23', 'wan22', 'seedance2')."
45
45
  },
46
46
  "music": {
47
47
  "type": "string",
@@ -116,8 +116,8 @@
116
116
  "type": "object",
117
117
  "additionalProperties": false,
118
118
  "properties": {
119
- "image": { "type": "string", "description": "Preferred image model (e.g., 'flux2', 'gpt-image-2')." },
120
- "video": { "type": "string", "description": "Preferred video model (e.g., 'ltx23', 'wan22', 'seedance2')." },
119
+ "image": { "type": "string", "description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')." },
120
+ "video": { "type": "string", "description": "Preferred video model (e.g., 'ltx25', 'ltx23', 'wan22', 'seedance2')." },
121
121
  "music": { "type": "string", "description": "Preferred music model." }
122
122
  }
123
123
  },
@@ -1,9 +1,9 @@
1
1
  {
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
- "$id": "https://schemas.sogni.ai/creative-agent/2026-04-27.1/tools/edit_image.schema.json",
3
+ "$id": "https://schemas.sogni.ai/creative-agent/2026-07-18.1/tools/edit_image.schema.json",
4
4
  "title": "edit_image arguments",
5
- "schemaVersion": "2026-04-27.1",
6
- "description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images, Flux.2 up to 6 images, and Qwen up to 3 images. Best for style-guided generation, combining elements from multiple images, ANY persona image creation, and any uploaded brand asset reuse — logos, brand marks, mascots, product shots, photos, screenshots, sketches, or character designs the user expects to appear in or guide the result. ALWAYS use this (never generate_image) when persona photos OR uploaded image assets meant for reuse are in context — even if a specific model is requested. Exception: explicit Z-image/Z-image Turbo uploaded-image enhancement uses generate_image with sourceImageIndex and starting_image_strength because edit_image does not expose Z-image models. If a previous edit_image attempt did not preserve the uploaded asset well, stay on edit_image and tighten the prompt or switch model — do not fall back to generate_image, which has no access to the upload at all. For direct edits (remove objects, enhance), use restore_photo or refine_result unless the user explicitly requested Z-image.",
5
+ "schemaVersion": "2026-07-18.1",
6
+ "description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images, Qwen up to 3 images, and Krea 2 Identity Edit / Dark Beast Krea 2 Identity Edit up to 2 images. Best for style-guided generation, combining elements from multiple images, ANY persona image creation, identity-preserving Krea edits, and any uploaded brand asset reuse — logos, brand marks, mascots, product shots, photos, screenshots, sketches, or character designs the user expects to appear in or guide the result. ALWAYS use this (never generate_image) when persona photos OR uploaded image assets meant for reuse are in context — even if a specific edit model is requested. Exception: explicit Z-image/Z-image Turbo/Krea 2 Turbo uploaded-image enhancement uses generate_image with sourceImageIndex and starting_image_strength because those base image-to-image models are not edit_image models. If a previous edit_image attempt did not preserve the uploaded asset well, stay on edit_image and tighten the prompt or switch model; generate_image has no access to the upload.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "properties": {
@@ -17,9 +17,10 @@
17
17
  "gpt-image-2",
18
18
  "qwen-lightning",
19
19
  "qwen",
20
- "flux2"
20
+ "krea-identity-edit",
21
+ "dark-beast-krea2-identity-edit"
21
22
  ],
22
- "description": "DO NOT SET THIS PARAMETER unless the user names a specific edit model, asks for a very complex reference-guided image render, or asks for a video storyboard/storyboard sheet/contact sheet/panel layout image using references. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 reference-guided image/edit/model, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Z-image and Z-image Turbo are not edit_image models; route those explicit uploaded-image enhancement requests to generate_image with sourceImageIndex and starting_image_strength. If the user names another edit/image model, honor that requested model instead. GPT Image 2 always processes input images at high fidelity; do not set input_fidelity."
23
+ "description": "The app auto-selects Fast→Qwen Lightning and HQ/Pro→full Qwen only for ordinary identity-neutral edits. REQUIRED IDENTITY DEFAULT: set \"krea-identity-edit\" whenever an edit of a referenced person or character must keep likeness or character identity while changing clothing, hair or makeup, pose or position, face/head/body, background, lighting, or visual style. Infer that semantic intent in any language; never route from keyword or regex matching. Also use it for a non-Pro single-character sheet. This default applies even when the user did not name Krea; an explicitly requested model always wins. Set \"dark-beast-krea2-identity-edit\" only when the user explicitly requests that model, its uncensored/community variant, or dark_beast_krea2_identity_edit_v1_2. Set \"gpt-image-2\" when the user explicitly names GPT/OpenAI/ChatGPT Image, or when precise typography, dense labels, or a professional multi-panel layout is the primary requirement; Pro character sheets may retain GPT Image 2. If GPT Image 2 is unavailable for detail-critical layout work, fall back to full \"qwen\", never \"qwen-lightning\". Krea identity edit models require at least one reference image, accept up to two context images, and work best at 512-2048px. Let the model tier and worker choose current steps, guidance, sampler, scheduler, grounding, and reference-boost defaults; do not send a negative prompt. When Krea is selected, override the generic prompt-length guidance with a concise 1-4 sentence delta instruction; name only the requested change and details that must remain fixed. Put the base scene/image first and an optional person/detail reference second. Z-image, Z-image Turbo, and base Krea 2 Turbo are generate_image img2img models, not edit_image selectors. If the user names another edit/image model, honor it. GPT Image 2 always processes input images at high fidelity; do not set input_fidelity."
23
24
  },
24
25
  "sourceImageIndex": {
25
26
  "type": "number",
@@ -33,11 +34,11 @@
33
34
  },
34
35
  "width": {
35
36
  "type": "number",
36
- "description": "Output image width in pixels. Defaults to the context image width. Supported range is 256-2560 for Qwen/Flux.2 edit models. For gpt-image-2, dimensions are flexible up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400; the renderer snaps to the nearest valid multiple-of-16 size. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
37
+ "description": "Output image width in pixels. Defaults to the context image width. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
37
38
  },
38
39
  "height": {
39
40
  "type": "number",
40
- "description": "Output image height in pixels. Defaults to the context image height. Supported range is 256-2560 for Qwen/Flux.2 edit models. For gpt-image-2, dimensions are flexible up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400; the renderer snaps to the nearest valid multiple-of-16 size. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
41
+ "description": "Output image height in pixels. Defaults to the context image height. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
41
42
  },
42
43
  "aspectRatio": {
43
44
  "type": "string",
@@ -17,7 +17,7 @@
17
17
  },
18
18
  "destination_model": {
19
19
  "type": "string",
20
- "description": "Optional destination model selector, such as seedance2, ltx23, wan22, flux2, gpt-image-2, or sdxl."
20
+ "description": "Optional destination model selector, such as seedance2, ltx25, ltx23, wan22, gpt-image-2, or sdxl."
21
21
  },
22
22
  "destination_tool": {
23
23
  "type": "string",
@@ -3,7 +3,7 @@
3
3
  "$id": "https://schemas.sogni.ai/creative-agent/2026-04-27.1/tools/extend_video.schema.json",
4
4
  "title": "extend_video arguments",
5
5
  "schemaVersion": "2026-04-27.1",
6
- "description": "Extend a video by adding new time to the end. Works on BOTH videos previously rendered in this session AND user-uploaded videos — set videoIndex to a negative number (e.g. -1) to target an uploaded video when no prior render exists. The base video is auto-selected from the most recent video in this session unless videoIndex is set. For LTX-2.3 base clips, the tool extracts the last frame and renders an image-to-video continuation. For Seedance base clips, the tool extracts a trailing reference segment and renders a video-to-video continuation. Returns both the standalone new segment and a spliced composite (base + new segment). Use when the user asks to \"make it longer\", \"extend the video\", \"add another N seconds\", \"continue the scene\", \"add an outro/bumper to the end\", etc. Prefer this over generate_image+animate_photo+stitch_video for \"add a bumper/outro to this video\" — extend_video preserves the original base bytes, audio, and timing instead of re-encoding them. Do not use this tool to render fresh videos from scratch — call generate_video or animate_photo for that. Output durations follow each model's native limits (LTX 2-20s, Seedance 4-15s) for the new segment alone.",
6
+ "description": "Extend a video by adding new time to the end. For LTX 2.5/2.3 base clips, the tool extracts the last frame and renders an image-to-video continuation; new non-Seedance continuations default to LTX 2.5. For Seedance base clips, it preserves the Seedance path. Returns both the standalone new segment and the spliced composite while preserving the original base bytes, audio, and timing.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "properties": {
@@ -13,9 +13,9 @@
13
13
  },
14
14
  "duration": {
15
15
  "type": "number",
16
- "description": "Length in seconds of the new appended segment (NOT total final length). LTX 2-20, Seedance 4-15. Default: 5.",
16
+ "description": "Length in seconds of the new appended segment (NOT total final length). LTX 2-20, Seedance 2.0/Mini/Fast 4-15, Seedance 2.5 4-30. Default: 5.",
17
17
  "minimum": 2,
18
- "maximum": 20
18
+ "maximum": 30
19
19
  },
20
20
  "videoIndex": {
21
21
  "type": "number",
@@ -25,11 +25,14 @@
25
25
  "type": "string",
26
26
  "enum": [
27
27
  "auto",
28
+ "ltx25",
28
29
  "ltx23",
29
30
  "seedance2",
30
- "seedance2-fast"
31
+ "seedance2-mini",
32
+ "seedance2-fast",
33
+ "seedance2-5"
31
34
  ],
32
- "description": "Which model to use for the new segment. Default: \"auto\" — detect from the base video's producer (Seedance base → Seedance, otherwise LTX-2.3). Override only when the user explicitly requests a different model."
35
+ "description": "Which model to use for the new segment. Default: \"auto\" — preserve Seedance for a Seedance base and otherwise use LTX 2.5. Use ltx23 only for explicit rollback. \"seedance2-5\" supports 480p/720p and 4-30s of new footage at 24 fps."
33
36
  },
34
37
  "keepOriginalAudio": {
35
38
  "type": "boolean",
@@ -1,9 +1,9 @@
1
1
  {
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
- "$id": "https://schemas.sogni.ai/creative-agent/2026-04-27.1/tools/generate_image.schema.json",
3
+ "$id": "https://schemas.sogni.ai/creative-agent/2026-07-18.1/tools/generate_image.schema.json",
4
4
  "title": "generate_image arguments",
5
- "schemaVersion": "2026-04-27.1",
6
- "description": "Generate a new image from a text description. Usually this is text-only: do NOT use this tool when the user expects an existing image to be reused or preserved in the result. That includes (a) people from My Personas, and (b) uploaded assets such as logos, brand marks, mascots, product shots, photos, screenshots, sketches, character designs, or other reference images they want carried through. Use edit_image with sourceImageIndex=-1 (or the appropriate generated index) instead. Exception: when the user explicitly requests Z-image, Z Image, or Z-image Turbo for an uploaded-image enhancement/image-to-image request, use this tool with model=\"z-turbo\" or model=\"z-image\", sourceImageIndex=-1, and starting_image_strength because edit_image does not expose Z-image models.",
5
+ "schemaVersion": "2026-07-18.1",
6
+ "description": "Generate a new image from a text description. Usually this is text-only: do NOT use this tool when the user expects an existing image to be reused or preserved in the result. That includes (a) people from My Personas, and (b) uploaded assets such as logos, brand marks, mascots, product shots, photos, screenshots, sketches, character designs, or other reference images they want carried through. Use edit_image with sourceImageIndex=-1 (or the appropriate generated index) instead. Exception: when the user explicitly requests Z-image, Z Image, Z-image Turbo, or Krea 2 Turbo for an uploaded-image enhancement/image-to-image request, use this tool with model=\"z-turbo\", model=\"z-image\", or model=\"krea-2-turbo\", sourceImageIndex=-1, and starting_image_strength because edit_image does not expose those base image-to-image models.",
7
7
  "type": "object",
8
8
  "additionalProperties": false,
9
9
  "properties": {
@@ -17,15 +17,18 @@
17
17
  "gpt-image-2",
18
18
  "z-turbo",
19
19
  "z-image",
20
+ "krea-2-turbo",
21
+ "dark-beast-krea2",
22
+ "dark-beast-z-turbo",
20
23
  "chroma-v46-flash",
24
+ "chroma1-hd",
21
25
  "chroma-detail",
22
- "flux1-krea",
23
- "flux2",
24
26
  "pony-v7",
25
27
  "qwen-2512",
26
28
  "qwen-2512-lightning",
27
29
  "albedo-xl",
28
30
  "animagine-xl",
31
+ "one-obsession-v22",
29
32
  "anima-pencil-xl",
30
33
  "art-universe-xl",
31
34
  "hyphoria-real",
@@ -37,15 +40,15 @@
37
40
  "pony-faetality",
38
41
  "dreamshaper-xl"
39
42
  ],
40
- "description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, or explicitly asks for Z-image/Z-image Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"/\"flux2\"/\"flux1-krea\" CANNOT do nudity — use \"pony-v7\", \"chroma-detail\", \"chroma-v46-flash\", or \"z-turbo\" instead."
43
+ "description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, asks for anime without naming a model, requests permitted NSFW/nudity content, or explicitly asks for Z-image/Z-image Turbo/Krea 2 Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"one-obsession-v22\" when the user asks for an anime or anime-style image and has not named a specific image model. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. Set \"krea-2-turbo\" when the user asks for Krea 2 Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image and Krea 2 Turbo image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"/Qwen image models CANNOT do nudity. For permitted NSFW/nudity content, prefer \"dark-beast-krea2\", then \"dark-beast-z-turbo\"; \"chroma1-hd\", \"pony-v7\", \"chroma-detail\", \"chroma-v46-flash\", and \"z-turbo\" are compatible fallbacks."
41
44
  },
42
45
  "width": {
43
46
  "type": "number",
44
- "description": "Output image width in pixels. Default: 1024. Supported range is 256-2560 for default Z/Qwen/Flux.2 image models and 256-2048 for legacy/specialized image models. For gpt-image-2, dimensions are flexible up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400; the renderer snaps to the nearest valid multiple-of-16 size. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"1080x1920\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds, so do not ask the user to adjust by a few pixels."
47
+ "description": "Output image width in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"1080x1920\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
45
48
  },
46
49
  "height": {
47
50
  "type": "number",
48
- "description": "Output image height in pixels. Default: 1024. Supported range is 256-2560 for default Z/Qwen/Flux.2 image models and 256-2048 for legacy/specialized image models. For gpt-image-2, dimensions are flexible up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400; the renderer snaps to the nearest valid multiple-of-16 size. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"1080x1920\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds, so do not ask the user to adjust by a few pixels."
51
+ "description": "Output image height in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"1080x1920\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
49
52
  },
50
53
  "numberOfVariations": {
51
54
  "type": "number",
@@ -59,7 +62,7 @@
59
62
  },
60
63
  "starting_image_strength": {
61
64
  "type": "number",
62
- "description": "Image-to-image strength (0.0-1.0). Only used when a source image is available and model supports img2img. Higher values = more deviation from the source image. 0.35 = conservative enhancement, 0.5 = balanced, 0.8 = creative. Set this with sourceImageIndex when the user explicitly requests Z-image/Z-image Turbo enhancement or any supported img2img starting-image workflow."
65
+ "description": "Image-to-image source guidance strength (0.0-1.0). Only set when a source image is available and the model supports img2img. For Z-Image/Z-Image Turbo, Krea 2 Turbo, or source-preserving enhancement requests, use 0.75 with sourceImageIndex so the source image remains a strong guide while allowing higher-resolution reconstruction. Use lower values only when the user explicitly asks for a lighter guide/subtle variation; higher values are more creative and can deviate further from the source."
63
66
  },
64
67
  "sourceImageIndex": {
65
68
  "type": "number",
@@ -35,9 +35,10 @@
35
35
  "type": "string",
36
36
  "enum": [
37
37
  "turbo",
38
- "sft"
38
+ "sft",
39
+ "music3"
39
40
  ],
40
- "description": "ACE-Step model variant. \"turbo\" (default): Higher quality audio generation with 4-16 steps and half the cost. Always use turbo unless the user explicitly requests the SFT model. \"sft\": Experimental model with lower audio quality but very strong lyric handling. 10-200 steps, full cost. Only use when the user specifically asks for SFT. Default: \"turbo\"."
41
+ "description": "Music model. \"turbo\" (default): ACE-Step 1.5 Turbo — fast 4-16 step drafts at half cost. \"sft\": ACE-Step 1.5 SFT — experimental, strong lyric handling, 10-200 steps, full cost. \"music3\": MiniMax Music 3 — premium autoregressive composer with the best vocals, lyric adherence and song structure; 30 steps, up to 5 minutes, ~20x turbo cost, and it treats duration as a ceiling (may end the song early at a musical resolution). Use music3 when the user asks for the best quality, realistic vocals, or full songs; otherwise default to \"turbo\"."
41
42
  },
42
43
  "timesig": {
43
44
  "type": "number",