@sogni-ai/sogni-intelligence-client 3.15.0 → 3.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/dist/client/SogniClientWrapper.d.ts.map +1 -1
  2. package/dist/client/SogniClientWrapper.js +25 -22
  3. package/dist/client/SogniClientWrapper.js.map +1 -1
  4. package/dist/contracts/hostedComposition.d.ts +4 -0
  5. package/dist/contracts/hostedComposition.d.ts.map +1 -1
  6. package/dist/contracts/hostedComposition.js +4 -0
  7. package/dist/contracts/hostedComposition.js.map +1 -1
  8. package/dist/contracts/index.d.ts +1 -1
  9. package/dist/contracts/index.d.ts.map +1 -1
  10. package/dist/contracts/index.js.map +1 -1
  11. package/dist/contracts/musicComposition.d.ts +8 -2
  12. package/dist/contracts/musicComposition.d.ts.map +1 -1
  13. package/dist/contracts/musicComposition.js +60 -4
  14. package/dist/contracts/musicComposition.js.map +1 -1
  15. package/dist/index.d.ts +2 -1
  16. package/dist/index.d.ts.map +1 -1
  17. package/dist/index.js +3 -2
  18. package/dist/index.js.map +1 -1
  19. package/dist/media/musicSettings.d.ts +39 -2
  20. package/dist/media/musicSettings.d.ts.map +1 -1
  21. package/dist/media/musicSettings.js +13 -0
  22. package/dist/media/musicSettings.js.map +1 -1
  23. package/dist/media/videoSettings.d.ts +1 -1
  24. package/dist/media/videoSettings.d.ts.map +1 -1
  25. package/dist/media/videoSettings.js +20 -0
  26. package/dist/media/videoSettings.js.map +1 -1
  27. package/dist/openai-tools/_manifests.generated.d.ts.map +1 -1
  28. package/dist/openai-tools/_manifests.generated.js +51 -41
  29. package/dist/openai-tools/_manifests.generated.js.map +1 -1
  30. package/dist/openai-tools/composition-tools.json +3 -3
  31. package/dist/openai-tools/generation-tools.json +48 -38
  32. package/dist/schemas/tools/animate_photo.schema.json +7 -6
  33. package/dist/schemas/tools/compose_script.schema.json +1 -1
  34. package/dist/schemas/tools/compose_workflow.schema.json +2 -2
  35. package/dist/schemas/tools/compose_workflow_template.schema.json +2 -2
  36. package/dist/schemas/tools/edit_image.schema.json +4 -5
  37. package/dist/schemas/tools/enhance_prompt.schema.json +1 -1
  38. package/dist/schemas/tools/extend_video.schema.json +3 -2
  39. package/dist/schemas/tools/generate_image.schema.json +3 -5
  40. package/dist/schemas/tools/generate_music.schema.json +3 -2
  41. package/dist/schemas/tools/generate_video.schema.json +10 -8
  42. package/dist/schemas/tools/replace_video_segment.schema.json +2 -1
  43. package/dist/schemas/tools/sound_to_video.schema.json +11 -5
  44. package/dist/schemas/tools/video_to_video.schema.json +5 -4
  45. package/dist/tools/definitions/generate-music/definition.d.ts.map +1 -1
  46. package/dist/tools/definitions/generate-music/definition.js +5 -3
  47. package/dist/tools/definitions/generate-music/definition.js.map +1 -1
  48. package/dist/tools/definitions/generate-video/definition.d.ts.map +1 -1
  49. package/dist/tools/definitions/generate-video/definition.js +2 -1
  50. package/dist/tools/definitions/generate-video/definition.js.map +1 -1
  51. package/dist/tools/shared/modelRegistry.d.ts.map +1 -1
  52. package/dist/tools/shared/modelRegistry.js +2 -0
  53. package/dist/tools/shared/modelRegistry.js.map +1 -1
  54. package/dist/utils/helpers.d.ts +6 -0
  55. package/dist/utils/helpers.d.ts.map +1 -1
  56. package/dist/utils/helpers.js +15 -0
  57. package/dist/utils/helpers.js.map +1 -1
  58. package/dist-esm/client/SogniClientWrapper.js +26 -23
  59. package/dist-esm/client/SogniClientWrapper.js.map +1 -1
  60. package/dist-esm/contracts/hostedComposition.js +4 -0
  61. package/dist-esm/contracts/hostedComposition.js.map +1 -1
  62. package/dist-esm/contracts/index.js.map +1 -1
  63. package/dist-esm/contracts/musicComposition.js +60 -4
  64. package/dist-esm/contracts/musicComposition.js.map +1 -1
  65. package/dist-esm/index.js +1 -1
  66. package/dist-esm/index.js.map +1 -1
  67. package/dist-esm/media/musicSettings.js +13 -0
  68. package/dist-esm/media/musicSettings.js.map +1 -1
  69. package/dist-esm/media/videoSettings.js +20 -0
  70. package/dist-esm/media/videoSettings.js.map +1 -1
  71. package/dist-esm/openai-tools/_manifests.generated.js +51 -41
  72. package/dist-esm/openai-tools/_manifests.generated.js.map +1 -1
  73. package/dist-esm/openai-tools/composition-tools.json +3 -3
  74. package/dist-esm/openai-tools/generation-tools.json +48 -38
  75. package/dist-esm/schemas/tools/animate_photo.schema.json +7 -6
  76. package/dist-esm/schemas/tools/compose_script.schema.json +1 -1
  77. package/dist-esm/schemas/tools/compose_workflow.schema.json +2 -2
  78. package/dist-esm/schemas/tools/compose_workflow_template.schema.json +2 -2
  79. package/dist-esm/schemas/tools/edit_image.schema.json +4 -5
  80. package/dist-esm/schemas/tools/enhance_prompt.schema.json +1 -1
  81. package/dist-esm/schemas/tools/extend_video.schema.json +3 -2
  82. package/dist-esm/schemas/tools/generate_image.schema.json +3 -5
  83. package/dist-esm/schemas/tools/generate_music.schema.json +3 -2
  84. package/dist-esm/schemas/tools/generate_video.schema.json +10 -8
  85. package/dist-esm/schemas/tools/replace_video_segment.schema.json +2 -1
  86. package/dist-esm/schemas/tools/sound_to_video.schema.json +11 -5
  87. package/dist-esm/schemas/tools/video_to_video.schema.json +5 -4
  88. package/dist-esm/tools/definitions/generate-music/definition.js +5 -3
  89. package/dist-esm/tools/definitions/generate-music/definition.js.map +1 -1
  90. package/dist-esm/tools/definitions/generate-video/definition.js +2 -1
  91. package/dist-esm/tools/definitions/generate-video/definition.js.map +1 -1
  92. package/dist-esm/tools/shared/modelRegistry.js +2 -0
  93. package/dist-esm/tools/shared/modelRegistry.js.map +1 -1
  94. package/dist-esm/utils/helpers.js +14 -0
  95. package/dist-esm/utils/helpers.js.map +1 -1
  96. package/package.json +3 -3
@@ -320,7 +320,7 @@ exports.compositionToolsManifest = {
320
320
  "properties": {
321
321
  "prompt": { "type": "string", "description": "The source prompt, rough idea, or prompt revision request to enhance." },
322
322
  "target_output": { "type": "string", "enum": ["image_prompt", "video_prompt", "music_prompt", "edit_prompt", "model_prompt", "general_prompt"], "description": "The kind of prompt artifact to produce." },
323
- "destination_model": { "type": "string", "description": "Optional destination model selector, such as seedance2, ltx23, wan22, flux2, gpt-image-2, or sdxl." },
323
+ "destination_model": { "type": "string", "description": "Optional destination model selector, such as seedance2, ltx23, wan22, gpt-image-2, or sdxl." },
324
324
  "destination_tool": { "type": "string", "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, video_to_video, or generate_music." },
325
325
  "prompting_type": { "type": "string", "enum": ["flux", "sdxl", "sd15", "pony", "fast", "sd3", "editing", "video"], "description": "Optional image-prompting family when producing an image prompt." },
326
326
  "model_title": { "type": "string", "description": "Optional human-readable target model name for image prompt guidance." },
@@ -431,7 +431,7 @@ exports.compositionToolsManifest = {
431
431
  "type": "object",
432
432
  "additionalProperties": false,
433
433
  "properties": {
434
- "image": { "type": "string", "description": "Preferred image model (e.g., 'flux2', 'gpt-image-2')." },
434
+ "image": { "type": "string", "description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')." },
435
435
  "video": { "type": "string", "description": "Preferred video model (e.g., 'ltx23', 'wan22', 'seedance2')." },
436
436
  "music": { "type": "string", "description": "Preferred music model." }
437
437
  }
@@ -507,7 +507,7 @@ exports.compositionToolsManifest = {
507
507
  "type": "object",
508
508
  "additionalProperties": false,
509
509
  "properties": {
510
- "image": { "type": "string", "description": "Preferred image model (e.g., 'flux2', 'gpt-image-2')." },
510
+ "image": { "type": "string", "description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')." },
511
511
  "video": { "type": "string", "description": "Preferred video model (e.g., 'ltx23', 'wan22', 'seedance2')." },
512
512
  "music": { "type": "string", "description": "Preferred music model." }
513
513
  }
@@ -527,7 +527,7 @@ exports.compositionToolsManifest = {
527
527
  ]
528
528
  };
529
529
  exports.generationToolsManifest = {
530
- "version": "2026-04-27.1",
530
+ "version": "2026-08-14.1",
531
531
  "source": "sogni-creative-agent/src/tools/definitions/*/definition.ts",
532
532
  "schemaRefs": {
533
533
  "generate_image": "../schemas/tools/generate_image.schema.json",
@@ -574,8 +574,6 @@ exports.generationToolsManifest = {
574
574
  "chroma-v46-flash",
575
575
  "chroma1-hd",
576
576
  "chroma-detail",
577
- "flux1-krea",
578
- "flux2",
579
577
  "pony-v7",
580
578
  "qwen-2512",
581
579
  "qwen-2512-lightning",
@@ -593,15 +591,15 @@ exports.generationToolsManifest = {
593
591
  "pony-faetality",
594
592
  "dreamshaper-xl"
595
593
  ],
596
- "description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, asks for anime without naming a model, requests permitted NSFW/nudity content, or explicitly asks for Z-image/Z-image Turbo/Krea 2 Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"one-obsession-v22\" when the user asks for an anime or anime-style image and has not named a specific image model. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. Set \"krea-2-turbo\" when the user asks for Krea 2 Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image and Krea 2 Turbo image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"/\"flux2\"/\"flux1-krea\"/Qwen image models CANNOT do nudity. For permitted NSFW/nudity content, prefer \"dark-beast-krea2\", then \"dark-beast-z-turbo\"; \"chroma1-hd\", \"pony-v7\", \"chroma-detail\", \"chroma-v46-flash\", and \"z-turbo\" are compatible fallbacks."
594
+ "description": "DO NOT SET THIS PARAMETER unless the user names a specific model, asks for a very complex image render, asks for a video storyboard/storyboard sheet/contact sheet/panel layout image, asks for anime without naming a model, requests permitted NSFW/nudity content, or explicitly asks for Z-image/Z-image Turbo/Krea 2 Turbo image-to-image. The app auto-selects based on quality settings. Set \"gpt-image-2\" when the user asks for a ChatGPT, OpenAI, GPT, GPT-2, GPT Image, or gpt-image-2 image/model, when they explicitly request very strong text rendering, or by default for complex single-image renders that need dense labels, crisp typography, multi-panel composition, timing notes, foley notes, professional storyboard-sheet layout, or a comprehensive character/mascot/model sheet with turnarounds, expressions, accessories, palette swatches, and brand notes. Set \"one-obsession-v22\" when the user asks for an anime or anime-style image and has not named a specific image model. Set \"z-turbo\" when the user asks for Z-image Turbo; set \"z-image\" when they ask for Z-image without Turbo. Set \"krea-2-turbo\" when the user asks for Krea 2 Turbo. If the user names another image model, honor that requested model instead. A model preference usually does not change which tool to use; the Z-image and Krea 2 Turbo image-to-image exception uses sourceImageIndex plus starting_image_strength on this tool. NSFW rule: \"gpt-image-2\"/Qwen image models CANNOT do nudity. For permitted NSFW/nudity content, prefer \"dark-beast-krea2\", then \"dark-beast-z-turbo\"; \"chroma1-hd\", \"pony-v7\", \"chroma-detail\", \"chroma-v46-flash\", and \"z-turbo\" are compatible fallbacks."
597
595
  },
598
596
  "width": {
599
597
  "type": "number",
600
- "description": "Output image width in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; Flux.2 uses a 2048x2048 total pixel budget (4,194,304 pixels) with non-square edges up to 2816, such as 1408x2816; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"1080x1920\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
598
+ "description": "Output image width in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"1080x1920\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
601
599
  },
602
600
  "height": {
603
601
  "type": "number",
604
- "description": "Output image height in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; Flux.2 uses a 2048x2048 total pixel budget (4,194,304 pixels) with non-square edges up to 2816, such as 1408x2816; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"1080x1920\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
602
+ "description": "Output image height in pixels. Default: 1024. Supported bounds depend on the selected image model: Z-Image/Z-Image Turbo, Dark Beast Z-Image Turbo, Chroma, and legacy/specialized image models support 256-2048 on either edge; Krea 2 Turbo, Dark Beast KREA 2, and Qwen image models support 256-2560 on either edge; One Obsession v22 supports 256-1920 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"1080x1920\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
605
603
  },
606
604
  "numberOfVariations": {
607
605
  "type": "number",
@@ -664,7 +662,7 @@ exports.generationToolsManifest = {
664
662
  "type": "function",
665
663
  "function": {
666
664
  "name": "generate_video",
667
- "description": "Generate a video from text or Seedance multimodal references. LTX 2.3 generates audio natively (dialogue, sounds, ambient music) — describe audio in the prompt. If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 (WAN 2.2 does not support voice identity). For non-Seedance syncing to a specific song or audio track, use sound_to_video instead. For non-Seedance animation from a locked source photo, use animate_photo. Do NOT use for My Personas unless generating a Seedance reference-based video — standard persona videos use resolve_personas → edit_image → animate_photo. SEEDANCE DEFAULT: For seedance2, seedance2-mini, seedance2-fast, or seedance2-5, default to exactly one video (4-15s on 2.0/Mini/Fast, 4-30s on seedance2-5) unless the user explicitly asks for multiple separate outputs. Multiple beats, shots, or scene descriptions in one Seedance prompt within the selected model's per-clip limit are still one video. If the user requests one continuous Seedance video longer than 15s, prefer \"seedance2-5\", which renders up to 30s in a single call; beyond 30s (or on 2.0/Mini/Fast) preserve the requested total duration in the prompt/context and let chat orchestration split it into supported segment renders and stitch them instead of clamping it to a short excerpt. Uploaded/generated storyboard, shot-sheet, or trailer-concept images used as Seedance references should become one Seedance generate_video call by default; do not extract panels with edit_image and do not animate the storyboard sheet with LTX unless the user explicitly asks for separate non-Seedance clips. Seedance loose image, video, and audio references go through this tool; do not use animate_photo sourceImageIndex/frameRole/endImageIndex for Seedance. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. If the uploaded audio is the primary sync target, lip-sync target, or requested as sound-to-video/audio-sync, use sound_to_video with videoModel=\"seedance2-mini\" instead of this tool unless the user asks for full Seedance. Use referenceAudioIndices here only when audio is a loose reference under an image/video-anchored Seedance shot. For Seedance, every image — first frame, last frame, or loose reference — is passed through referenceImageIndices (auto-uploaded as referenceImageUrls). Anchor frame intent in the prompt with @Image tags such as \"Use @Image1 as the opening shot reference. Begin the video with a composition, subject placement, lighting, mood, and camera framing that closely match @Image1.\" (or @Image2 as the final shot reference). For seamless-loop or \"first frame and last frame identical\" requests with a single uploaded image, anchor it explicitly as both: \"Use @Image1 as both the first frame and last frame so the video loops cleanly back to the opening composition.\" Assign each useful @Image/@Video/@Audio tag a role. APPROVED STORYBOARD PRODUCTION: When the user asks for a production workflow from an approved storyboard, the chat orchestrator should use the durable CampaignStoryboard contract: render the composite board, audit it, generate per-scene GPT Image 2 keyframes, then render Seedance scene clips and stitch them. Do not replace that with a generic storyboard-reference video unless the user asks for a fast draft. PARTIAL VIDEO EDITS: Do NOT call generate_video to re-render an existing rendered/uploaded video just to change part of it (the bumper, the intro, the end card, a single scene, the last few seconds, etc.). Use replace_video_segment for that — it preserves the unchanged portion, keeps the original audio outside the replaced window, and costs far less. Likewise use extend_video to add new time to the end without rewriting the rest. If the request is vague, ask about vision/mood/style first. Only call once you have clear creative intent.",
665
+ "description": "Generate a video from text or Seedance multimodal references. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively (dialogue, sounds, ambient music) — describe audio in the prompt. If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Never use placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. For non-Seedance syncing to a specific song or audio track, use sound_to_video instead. For non-Seedance animation from a locked source photo, use animate_photo. Do NOT use for My Personas unless generating a Seedance reference-based video — standard persona videos use resolve_personas → edit_image → animate_photo. SEEDANCE DEFAULT: For seedance2, seedance2-mini, seedance2-fast, or seedance2-5, default to exactly one video (4-15s on 2.0/Mini/Fast, 4-30s on seedance2-5) unless the user explicitly asks for multiple separate outputs. Multiple beats, shots, or scene descriptions in one Seedance prompt within the selected model's per-clip limit are still one video. If the user requests one continuous Seedance video longer than 15s, prefer \"seedance2-5\", which renders up to 30s in a single call; beyond 30s (or on 2.0/Mini/Fast) preserve the requested total duration in the prompt/context and let chat orchestration split it into supported segment renders and stitch them instead of clamping it to a short excerpt. Uploaded/generated storyboard, shot-sheet, or trailer-concept images used as Seedance references should become one Seedance generate_video call by default; do not extract panels with edit_image and do not animate the storyboard sheet with LTX unless the user explicitly asks for separate non-Seedance clips. Seedance loose image, video, and audio references go through this tool; do not use animate_photo sourceImageIndex/frameRole/endImageIndex for Seedance. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. If the uploaded audio is the primary sync target, lip-sync target, or requested as sound-to-video/audio-sync, use sound_to_video with videoModel=\"seedance2-mini\" instead of this tool unless the user asks for full Seedance. Use referenceAudioIndices here only when audio is a loose reference under an image/video-anchored Seedance shot. For Seedance, every image — first frame, last frame, or loose reference — is passed through referenceImageIndices (auto-uploaded as referenceImageUrls). Anchor frame intent in the prompt with @Image tags such as \"Use @Image1 as the opening shot reference. Begin the video with a composition, subject placement, lighting, mood, and camera framing that closely match @Image1.\" (or @Image2 as the final shot reference). For seamless-loop or \"first frame and last frame identical\" requests with a single uploaded image, anchor it explicitly as both: \"Use @Image1 as both the first frame and last frame so the video loops cleanly back to the opening composition.\" Assign each useful @Image/@Video/@Audio tag a role. APPROVED STORYBOARD PRODUCTION: When the user asks for a production workflow from an approved storyboard, the chat orchestrator should use the durable CampaignStoryboard contract: render the composite board, audit it, generate per-scene GPT Image 2 keyframes, then render Seedance scene clips and stitch them. Do not replace that with a generic storyboard-reference video unless the user asks for a fast draft. PARTIAL VIDEO EDITS: Do NOT call generate_video to re-render an existing rendered/uploaded video just to change part of it (the bumper, the intro, the end card, a single scene, the last few seconds, etc.). Use replace_video_segment for that — it preserves the unchanged portion, keeps the original audio outside the replaced window, and costs far less. Likewise use extend_video to add new time to the end without rewriting the rest. If the request is vague, ask about vision/mood/style first. Only call once you have clear creative intent.",
668
666
  "parameters": {
669
667
  "type": "object",
670
668
  "properties": {
@@ -682,17 +680,18 @@ exports.generationToolsManifest = {
682
680
  },
683
681
  "duration": {
684
682
  "type": "number",
685
- "description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.3 and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30 — only \"seedance2-5\" can use the 16-30s part of this range. MiniMax H3 is quantized to a 17-frame grid at a fixed 24 fps and renders 124-362 frames, so an H3 clip runs 5.17-15.08 seconds and a requested length outside that window snaps to the nearest valid H3 length. Use when the user explicitly requests a specific length.",
683
+ "description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.5, LTX 2.3, and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30 — only \"seedance2-5\" can use the 16-30s part of this range. MiniMax H3 is quantized to a 17-frame grid at a fixed 24 fps and renders 124-362 frames, so an H3 clip runs 5.17-15.08 seconds and a requested length outside that window snaps to the nearest valid H3 length. Use when the user explicitly requests a specific length.",
686
684
  "minimum": 2,
687
685
  "maximum": 30
688
686
  },
689
687
  "negativePrompt": {
690
688
  "type": "string",
691
- "description": "Advanced LTX/WAN only. Use this field only when the user explicitly asks to set a separate negative prompt. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt. Do not set for MiniMax H3, Seedance, or HappyHorse."
689
+ "description": "Advanced LTX 2.5/LTX 2.3/WAN only. All standard LTX 2.5 workflow IDs accept this separate negative prompt. Use this field only when the user explicitly asks to set one. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt. Do not set for MiniMax H3, Seedance, or HappyHorse."
692
690
  },
693
691
  "videoModel": {
694
692
  "type": "string",
695
693
  "enum": [
694
+ "ltx25",
696
695
  "ltx23",
697
696
  "wan22",
698
697
  "seedance2",
@@ -704,9 +703,10 @@ exports.generationToolsManifest = {
704
703
  "happyhorse-1.1-t2v",
705
704
  "happyhorse-1.1-i2v",
706
705
  "happyhorse-1.1-r2v",
707
- "minimax-h3-r2v"
706
+ "minimax-h3-r2v",
707
+ "minimax-h3-r2v-turbo"
708
708
  ],
709
- "description": "Video model. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step variant and Default Media Quality Pro uses the non-distilled dev variant. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx23\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. There is no H3 Turbo r2v selector. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\", a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" only when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft, Mini, or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0, Mini, and Fast accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with no more than 30 reference media files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
709
+ "description": "Video model. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx25\". HappyHorse 1.1 can be used here for \"happyhorse-1.1-t2v\" text-to-video, \"happyhorse-1.1-i2v\" with one uploaded/generated first-frame image via referenceImageIndices, or \"happyhorse-1.1-r2v\" with 1-9 image references. For a locked still image/source-frame animation, animate_photo with videoModel=\"happyhorse-1.1-i2v\" is also valid. HappyHorse supports 720p/1080p, 3-15s clips, native synchronized audio that is always on, image-only references, and no negativePrompt or generateAudio input. MiniMax H3 standard text-to-video uses \"minimax-h3-t2v\"; use \"minimax-h3-t2v-turbo\" for the 4-step Turbo tier. The Turbo image-to-video and first-to-last-frame selectors are \"minimax-h3-i2v-turbo\" and \"minimax-h3-flf2v-turbo\"; they keep H3's standard geometry, frame grid, and native-audio contract. MiniMax H3 Ref2VA Turbo uses \"minimax-h3-r2v-turbo\", the dedicated four-step Euler/simple R2V tier with a 960x544 upstream-aligned default. H3 renders 5.17-15.08s clips at a fixed 24 fps inside a 1344x768 pixel budget on a 32px grid, jointly generates its own stereo audio, takes no negativePrompt input, and supports generateAudio=false to return a video without an audio track. MiniMax H3 reference-to-video uses \"minimax-h3-r2v\" for standard quality or \"minimax-h3-r2v-turbo\" for the dedicated four-step Turbo LoRA, a separate ref2va checkpoint and the only H3 mode that takes loose references: up to 9 reference images, 3 reference videos (24 fps, 2-15s, each with an optional soundtrack) and 3 standalone audio tracks, no more than 12 reference files in total, passed with referenceImageIndices/referenceVideoIndices/referenceAudioIndices. At least one visual reference is required: one or more images and/or videos. A video can be the only visual input; audio cannot be the sole input. H3 r2v references are NOT locked frames — name them in the prompt with H3's own 1-based per-type labels <Picture 1>/<Video 1>/<Audio 1> and give every one an explicit job (identity, style, camera movement, voice character), stating which reference wins when two disagree. Use animate_photo with \"minimax-h3-i2v\" for a first-frame animation or \"minimax-h3-flf2v\" with frameRole=\"both\" for a first-to-last-frame transition. Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance draft iteration unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" only when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft, Mini, or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets. Seedance 2.0, Mini, and Fast accept images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total; Seedance 2.5 accepts images (up to 30), videos (up to 10), and audios (up to 10), with no more than 30 reference media files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
710
710
  },
711
711
  "generateAudio": {
712
712
  "type": "boolean",
@@ -735,11 +735,11 @@ exports.generationToolsManifest = {
735
735
  },
736
736
  "width": {
737
737
  "type": "number",
738
- "description": "Video width in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Default resolution depends on model and quality tier: LTX Fast about 720p and High/Pro about 1080p; WAN Fast uses 480p short side and High/Pro uses 720p short side. Set width only when the user specifies an exact width or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target, not an instruction to make landscape 1280x720. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override the default media quality. Mappings when orientation is explicit: 480p landscape=854x480, 480p portrait=480x854, 720p landscape=1280x720, 720p portrait=720x1280, 1080p landscape=1920x1080, 1080p portrait=1080x1920, 4K landscape=3840x2160. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
738
+ "description": "Video width in pixels. LTX 2.5 and LTX 2.3: 640-3840. WAN: 480-1536. Default resolution depends on model and quality tier: LTX Fast about 720p and High/Pro about 1080p; WAN Fast uses 480p short side and High/Pro uses 720p short side. Set width only when the user specifies an exact width or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target, not an instruction to make landscape 1280x720. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override the default media quality. Mappings when orientation is explicit: 480p landscape=854x480, 480p portrait=480x854, 720p landscape=1280x720, 720p portrait=720x1280, 1080p landscape=1920x1080, 1080p portrait=1080x1920, 4K landscape=3840x2160. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
739
739
  },
740
740
  "height": {
741
741
  "type": "number",
742
- "description": "Video height in pixels. LTX 2.3: 640-3840. WAN: 480-1536. Set height only when the user specifies an exact height or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target; do not convert it to landscape dimensions unless the user says landscape/horizontal/widescreen. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override Default Media Quality, including Pro. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
742
+ "description": "Video height in pixels. LTX 2.5 and LTX 2.3: 640-3840. WAN: 480-1536. Set height only when the user specifies an exact height or orientation-qualified exact pixels. A bare named resolution like \"720p resolution\" is a short-side target; do not convert it to landscape dimensions unless the user says landscape/horizontal/widescreen. If the user gives only one exact dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested exact dimensions override Default Media Quality, including Pro. Non-step values are accepted when in bounds; LTX snaps to the nearest 64px step and WAN snaps to the nearest 16px step internally, so do not ask the user to adjust by a few pixels."
743
743
  },
744
744
  "targetResolution": {
745
745
  "type": "number",
@@ -757,7 +757,7 @@ exports.generationToolsManifest = {
757
757
  },
758
758
  "voicePersonaName": {
759
759
  "type": "string",
760
- "description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one described in the video (e.g. \"David\" narrates a scene featuring Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23."
760
+ "description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one described in the video (e.g. \"David\" narrates a scene featuring Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
761
761
  }
762
762
  },
763
763
  "required": [
@@ -802,9 +802,10 @@ exports.generationToolsManifest = {
802
802
  "type": "string",
803
803
  "enum": [
804
804
  "turbo",
805
- "sft"
805
+ "sft",
806
+ "music3"
806
807
  ],
807
- "description": "ACE-Step model variant. \"turbo\" (default): Higher quality audio generation with 4-16 steps and half the cost. Always use turbo unless the user explicitly requests the SFT model. \"sft\": Experimental model with lower audio quality but very strong lyric handling. 10-200 steps, full cost. Only use when the user specifically asks for SFT. Default: \"turbo\"."
808
+ "description": "Music model. \"turbo\" (default): ACE-Step 1.5 Turbo — fast 4-16 step drafts at half cost. \"sft\": ACE-Step 1.5 SFT — experimental, strong lyric handling, 10-200 steps, full cost. \"music3\": MiniMax Music 3 — premium autoregressive composer with the best vocals, lyric adherence and song structure; 30 steps, up to 5 minutes, ~20x turbo cost, and it treats duration as a ceiling (may end the song early at a musical resolution). Use music3 when the user asks for the best quality, realistic vocals, or full songs; otherwise default to \"turbo\"."
808
809
  },
809
810
  "timesig": {
810
811
  "type": "number",
@@ -833,7 +834,7 @@ exports.generationToolsManifest = {
833
834
  "type": "function",
834
835
  "function": {
835
836
  "name": "edit_image",
836
- "description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images, Flux.2 up to 6 images, Qwen up to 3 images, and Krea 2 Identity Edit / Dark Beast Krea 2 Identity Edit up to 2 images. Best for style-guided generation, combining elements from multiple images, ANY persona image creation, identity-preserving Krea edits, and any uploaded brand asset reuse — logos, brand marks, mascots, product shots, photos, screenshots, sketches, or character designs the user expects to appear in or guide the result. ALWAYS use this (never generate_image) when persona photos OR uploaded image assets meant for reuse are in context — even if a specific edit model is requested. Exception: explicit Z-image/Z-image Turbo/Krea 2 Turbo uploaded-image enhancement uses generate_image with sourceImageIndex and starting_image_strength because those base image-to-image models are not edit_image models. If a previous edit_image attempt did not preserve the uploaded asset well, stay on edit_image and tighten the prompt or switch model; generate_image has no access to the upload.",
837
+ "description": "Generate images guided by reference photos. Supports GPT Image 2 up to 16 images, Qwen up to 3 images, and Krea 2 Identity Edit / Dark Beast Krea 2 Identity Edit up to 2 images. Best for style-guided generation, combining elements from multiple images, ANY persona image creation, identity-preserving Krea edits, and any uploaded brand asset reuse — logos, brand marks, mascots, product shots, photos, screenshots, sketches, or character designs the user expects to appear in or guide the result. ALWAYS use this (never generate_image) when persona photos OR uploaded image assets meant for reuse are in context — even if a specific edit model is requested. Exception: explicit Z-image/Z-image Turbo/Krea 2 Turbo uploaded-image enhancement uses generate_image with sourceImageIndex and starting_image_strength because those base image-to-image models are not edit_image models. If a previous edit_image attempt did not preserve the uploaded asset well, stay on edit_image and tighten the prompt or switch model; generate_image has no access to the upload.",
837
838
  "parameters": {
838
839
  "type": "object",
839
840
  "properties": {
@@ -847,11 +848,10 @@ exports.generationToolsManifest = {
847
848
  "gpt-image-2",
848
849
  "qwen-lightning",
849
850
  "qwen",
850
- "flux2",
851
851
  "krea-identity-edit",
852
852
  "dark-beast-krea2-identity-edit"
853
853
  ],
854
- "description": "The app auto-selects Fast→Qwen Lightning, HQ→full Qwen, and Pro→Flux.2 only for ordinary identity-neutral edits. REQUIRED IDENTITY DEFAULT: set \"krea-identity-edit\" whenever an edit of a referenced person or character must keep likeness or character identity while changing clothing, hair or makeup, pose or position, face/head/body, background, lighting, or visual style. Infer that semantic intent in any language; never route from keyword or regex matching. Also use it for a non-Pro single-character sheet. This default applies even when the user did not name Krea; an explicitly requested model always wins. Set \"dark-beast-krea2-identity-edit\" only when the user explicitly requests that model, its uncensored/community variant, or dark_beast_krea2_identity_edit_v1_2. Set \"gpt-image-2\" when the user explicitly names GPT/OpenAI/ChatGPT Image, or when precise typography, dense labels, or a professional multi-panel layout is the primary requirement; Pro character sheets may retain GPT Image 2. If GPT Image 2 is unavailable for detail-critical layout work, fall back to full \"qwen\", never \"qwen-lightning\". Krea identity edit models require at least one reference image, accept up to two context images, and work best at 512-2048px. Let the model tier and worker choose current steps, guidance, sampler, scheduler, grounding, and reference-boost defaults; do not send a negative prompt. When Krea is selected, override the generic prompt-length guidance with a concise 1-4 sentence delta instruction; name only the requested change and details that must remain fixed. Put the base scene/image first and an optional person/detail reference second. Z-image, Z-image Turbo, and base Krea 2 Turbo are generate_image img2img models, not edit_image selectors. If the user names another edit/image model, honor it. GPT Image 2 always processes input images at high fidelity; do not set input_fidelity."
854
+ "description": "The app auto-selects Fast→Qwen Lightning and HQ/Pro→full Qwen only for ordinary identity-neutral edits. REQUIRED IDENTITY DEFAULT: set \"krea-identity-edit\" whenever an edit of a referenced person or character must keep likeness or character identity while changing clothing, hair or makeup, pose or position, face/head/body, background, lighting, or visual style. Infer that semantic intent in any language; never route from keyword or regex matching. Also use it for a non-Pro single-character sheet. This default applies even when the user did not name Krea; an explicitly requested model always wins. Set \"dark-beast-krea2-identity-edit\" only when the user explicitly requests that model, its uncensored/community variant, or dark_beast_krea2_identity_edit_v1_2. Set \"gpt-image-2\" when the user explicitly names GPT/OpenAI/ChatGPT Image, or when precise typography, dense labels, or a professional multi-panel layout is the primary requirement; Pro character sheets may retain GPT Image 2. If GPT Image 2 is unavailable for detail-critical layout work, fall back to full \"qwen\", never \"qwen-lightning\". Krea identity edit models require at least one reference image, accept up to two context images, and work best at 512-2048px. Let the model tier and worker choose current steps, guidance, sampler, scheduler, grounding, and reference-boost defaults; do not send a negative prompt. When Krea is selected, override the generic prompt-length guidance with a concise 1-4 sentence delta instruction; name only the requested change and details that must remain fixed. Put the base scene/image first and an optional person/detail reference second. Z-image, Z-image Turbo, and base Krea 2 Turbo are generate_image img2img models, not edit_image selectors. If the user names another edit/image model, honor it. GPT Image 2 always processes input images at high fidelity; do not set input_fidelity."
855
855
  },
856
856
  "sourceImageIndex": {
857
857
  "type": "number",
@@ -865,11 +865,11 @@ exports.generationToolsManifest = {
865
865
  },
866
866
  "width": {
867
867
  "type": "number",
868
- "description": "Output image width in pixels. Defaults to the context image width. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; Flux.2 uses a 2048x2048 total pixel budget (4,194,304 pixels) with non-square edges up to 2816, such as 1408x2816; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
868
+ "description": "Output image width in pixels. Defaults to the context image width. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a width, exact pixel dimensions, or a named resolution (e.g., \"1280 wide\", \"1280x720\", \"720p\", \"3840x2160\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
869
869
  },
870
870
  "height": {
871
871
  "type": "number",
872
- "description": "Output image height in pixels. Defaults to the context image height. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; Flux.2 uses a 2048x2048 total pixel budget (4,194,304 pixels) with non-square edges up to 2816, such as 1408x2816; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
872
+ "description": "Output image height in pixels. Defaults to the context image height. Supported bounds depend on the selected edit model: Qwen edit models support 256-2560 on either edge; Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit work best from 512-2048 on either edge; GPT Image 2 supports flexible dimensions up to 3840px on either edge with max 3:1 aspect ratio and a total pixel budget from 655,360 to 8,294,400. Set when the user specifies a height, exact pixel dimensions, or a named resolution (e.g., \"720 high\", \"1280x720\", \"720p\", \"2160x3840\"). If the user gives only one dimension, set only that dimension and preserve/infer the sensible aspect ratio. User-requested dimensions override the default media quality, including Pro. Non-multiple-of-16 values are accepted when in bounds; the renderer snaps to the nearest supported size internally, so do not ask the user to adjust by a few pixels."
873
873
  },
874
874
  "aspectRatio": {
875
875
  "type": "string",
@@ -1040,13 +1040,13 @@ exports.generationToolsManifest = {
1040
1040
  "type": "function",
1041
1041
  "function": {
1042
1042
  "name": "animate_photo",
1043
- "description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.3 or WAN 2.2. Do NOT use this tool for seedance2, seedance2-mini, seedance2-fast, or seedance2-5. Seedance media references — including Seedance 2.5 first-and-last-frame requests — must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.3 generates audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Avoid placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because WAN 2.2 does not support voice identity. PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, prefer generating both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, then call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. If the user explicitly wants separately created frame assets, preserve that staged instruction while keeping indices correct. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Call once you have clear creative intent. N-VIDEOS PATTERN: Avoid sequential animate_photo calls for N outputs. For a single fixed source/end frame where only prompt text varies, use sourceImageIndex + numberOfVariations=N + one Dynamic Prompt branch in prompt so Sogni submits one project with multiple jobs. If the user explicitly asks for Dynamic Prompt or Dynamic Template syntax, prefer this one-project path whenever every output uses the same source/end frames and shared settings, even if they also ask to stitch the completed clips afterward. Use sourceImageIndices/prompts for multi-segment stitched non-Seedance video, different source/end assets, different audio windows, different durations/dimensions, isolated retry lifecycle, or other per-output parameters. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips on ltx23 (30s total = 2 clips × 15s) and 10-second clips on wan22 (60s total on wan22 = 6 clips × 10s; do NOT pick 4 clips × 15s on wan22 — the wan22 worker rejects clips longer than 10s). Multi-source flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image and only prompt text varies, use sourceImageIndex=-1, frameRole=\"both\" if requested, endImageIndex=-1 if requested, numberOfVariations=N, and one Dynamic Prompt branch in prompt. For a long scripted/dialogue/storyboard video from a single supplied/uploaded image where each segment needs isolated exact dialogue or per-segment wiring, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when source/end asset wiring or other per-output parameters differ, pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame.",
1043
+ "description": "Animate a photo into video with motion, audio, and dialogue using LTX 2.5 by default, LTX 2.3 as rollback, or WAN 2.2. Do NOT use this tool for seedance2, seedance2-mini, seedance2-fast, or seedance2-5. Seedance media references — including Seedance 2.5 first-and-last-frame requests — must go through generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text in the prompt; for seamless-loop Seedance requests with one uploaded image, the prompt should anchor it as both the first frame and last frame. LTX/WAN NOTE: uploaded audio files are not loose references for ltx23/wan22; use sound_to_video when uploaded audio is the primary sync target. DANCE REQUESTS (\"make them dance\", \"do the X dance\"): use dance_montage — NOT this tool. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively — describe dialogue and ambient sounds directly in the prompt (do NOT pre-generate audio for this tool). If the user provides exact speech, include it in double quotes; if they only imply speech, describe the performance and voice without inventing quoted words. Avoid placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". PERSONA VOICE: Only when the user explicitly asks to use/clone a registered persona voice clip, call resolve_personas first, then set voicePersonaName to select which persona's voice clip to use. Do not set voicePersonaName for ordinary character dialogue or inferred voices; describe those voices in the prompt for native LTX audio. For cross-persona narration (e.g. David narrates a video of Aleyna), resolve both personas and set voicePersonaName to the narrator only if that registered voice was requested. Persona voice requires ltx23 because LTX 2.5 has no compatible ID-LoRA and WAN 2.2 does not support voice identity. PERSONA PIPELINE: For persona videos, ensure an image of the persona exists before calling animate_photo. The standard pipeline is: resolve_personas → edit_image → animate_photo. If a suitable persona image already exists (user uploaded one, a prior edit_image/generate_image result, OR the user explicitly says to use the Persona image/reference photo directly), skip edit_image and animate directly. After resolve_personas, this tool can animate the injected persona image directly when that explicit direct-use instruction is given. Auto-uses the latest result image (from any prior tool) unless sourceImageIndex is set. Supports start-frame (default), end-frame, and start+end interpolation modes for LTX/WAN — ask the user which frame role their image should play if they mention \"end frame\", \"last frame\", or provide two images. FIRST+LAST FRAME WORKFLOW: When the user wants a non-Seedance video using two different scenes as start and end frames, prefer generating both images in a single generate_image/edit_image call with numberOfVariations=2 and Dynamic Prompts, then call animate_photo with frameRole=\"both\", sourceImageIndex=0, endImageIndex=1. If the user explicitly wants separately created frame assets, preserve that staged instruction while keeping indices correct. In frameRole=\"both\", the handler automatically inspects both images and upgrades the base prompt into a scene-aware smooth transition prompt, so your prompt should state the desired transition style, action, dialogue, and audio rather than trying to list every visible object. If the request is vague, analyze the image first and suggest 2-3 specific animation ideas tailored to what you see. Call once you have clear creative intent. N-VIDEOS PATTERN: Avoid sequential animate_photo calls for N outputs. For a single fixed source/end frame where only prompt text varies, use sourceImageIndex + numberOfVariations=N + one Dynamic Prompt branch in prompt so Sogni submits one project with multiple jobs. If the user explicitly asks for Dynamic Prompt or Dynamic Template syntax, prefer this one-project path whenever every output uses the same source/end frames and shared settings, even if they also ask to stitch the completed clips afterward. Use sourceImageIndices/prompts for multi-segment stitched non-Seedance video, different source/end assets, different audio windows, different durations/dimensions, isolated retry lifecycle, or other per-output parameters. sourceImageIndices supports up to 16 entries; there is NO 3-clip cap, so do not split one planned batch into \"first 3\" and \"remaining\" calls. For a dialogue-heavy total-duration request with no explicit per-clip duration, prefer 15-second clips on ltx23 (30s total = 2 clips × 15s) and 10-second clips on wan22 (60s total on wan22 = 6 clips × 10s; do NOT pick 4 clips × 15s on wan22 — the wan22 worker rejects clips longer than 10s). Multi-source flavors: (A) SHARED CONTENT — when all N clips have the same dialogue/motion but different source visuals (different scenes, outfits, environments, persona looks), first generate N distinct images via ONE edit_image/generate_image call with numberOfVariations=N + Dynamic Prompts {|}, then call animate_photo with sourceImageIndices=[start..start+N-1] and a single shared `prompt`. If all segments intentionally reuse the primary uploaded image and only prompt text varies, use sourceImageIndex=-1, frameRole=\"both\" if requested, endImageIndex=-1 if requested, numberOfVariations=N, and one Dynamic Prompt branch in prompt. For a long scripted/dialogue/storyboard video from a single supplied/uploaded image where each segment needs isolated exact dialogue or per-segment wiring, use sourceImageIndices=[-1,-1,...] and per-clip prompts. Only set frameRole=\"both\" and endImageIndex=-1 when the user explicitly says the same uploaded/source/original image should be both the first and last frame of every segment. If the user requests generated source images first, honor that image stage, then animate the generated result indices. When using generated scene keyframes and each clip should begin and end on its own scene image for stitching, call animate_photo with frameRole=\"both\" and sourceImageIndices=[start..end] but OMIT endImageIndex; do not set endImageIndex=-1 unless every source is the uploaded image. (B) PER-CLIP CONTENT — when source/end asset wiring or other per-output parameters differ, pass BOTH sourceImageIndices AND `prompts` (an array of N strings, one per clip) in the same single call. Each prompt must independently anchor the visible characters, scene action, camera, audio, exact screenplay-style speaker tags, and exact quoted dialogue for that segment. If you just wrote or displayed a script/table, copy the exact dialogue lines into the corresponding per-clip prompts; do not summarize them as speech activity. If using named speaker tags with any multi-person reference image or generated scene keyframe, include one explicit cast map in each prompt that binds each name to visible position, clothing, and props/actions, e.g. SPEAKER_A = left person holding a prop; SPEAKER_B = center person with tablet; SPEAKER_C = right person near table. Do not also describe the same people again as generic man/boy/girl/woman/character subjects. For screenplay, storyboard, commercial, series, or other longer-form tasks with recurring characters, preserve the same character names and repeated visual anchors in every per-clip prompt where each character appears. Use the standard single-source path (numberOfVariations only) when the user wants motion variety from a single fixed frame.",
1044
1044
  "parameters": {
1045
1045
  "type": "object",
1046
1046
  "properties": {
1047
1047
  "prompt": {
1048
1048
  "type": "string",
1049
- "description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.3 generates audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same source/end frames and generation parameters and only prompt text varies. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
1049
+ "description": "I2V RULE: Do NOT re-describe what is visible in the input image. Focus on the transition from stillness — motion, expression changes, what happens next, camera movement, and sound.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. Set skipPromptProcessing=true; for Seedance also set expandPrompt=false.\n\nSTRUCTURE: \"[How the subject begins to move]. [What changes next]. [Camera behavior]. [Audio].\"\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main action beat + 1 simple camera move. Around 10s: 2-3 clear action beats + 1 camera move. >10s: up to 4 action beats in clear sequence. Prefer fewer readable beats over dense micro-actions, especially in short clips.\n\nBLOCKING: Use the image as the anchor and direct only meaningful layout changes. If the prompt introduces multiple moving subjects, state left/right placement, foreground/background, facing toward/away, and relative distance.\n\nACTION: One flowing paragraph. Describe motion beat by beat with temporal connectors (\"as\", \"then\", \"while\"). Specify who moves, what moves, how it moves, and what the camera does. One main thread — avoid too many actions at once or generic phrases like \"comes alive.\"\n\nDIALOGUE: Put user-provided spoken lines in double quotes. For screenplay-style or longer-form tasks, prefix each spoken line with a stable speaker tag outside the quotes, e.g. CHARACTER: \"We made it.\" Break long speech into short quoted phrases with acting beats between them (gestures, pauses, glances). If the user asks for speech but provides no exact words, describe the visible delivery, voice quality, and emotion without inventing quoted dialogue; ask only when exact wording is the point of the request. Never write placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\", or \"final line lands\". Show emotion through visible behavior, not labels. LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes (e.g. bold white text reading CONGRATULATIONS overlays the lower third).\n\nAUDIO: Prompt sound intentionally — voice quality, volume, room tone, ambience, music, weather, footsteps. Include language or accent if relevant. Useful voice/volume anchors: whisper, mutter, shout, scream, energetic announcer, resonant voice with gravitas, distorted radio-style, robotic monotone, childlike curiosity.\n\nCAMERA: Cinematic terms — slow push-in, static tripod, handheld, slow arc, dolly in. Describe movement relative to subject.\n\nFor first+last-frame transitions (frameRole=\"both\"), write a concise base request for the transition style, action, dialogue, and audio. The handler will inspect both frames and expand it into a scene-aware prompt that maps visible objects and subjects between frames.\n\nFor specific characters (movies, TV): describe visual appearance — don't rely on names alone.\n\nFor complex/creative scenes (characters talking, skits), capture full creative intent — system auto-expands into detailed prompt.\n\nAVOID: Re-describing the image, vague prompts, too many actions at once, abstract emotions without visible behavior, rigid numeric constraints, readable text or logos.\n\nWAN 2.2 (\"wan22\"): 30-150 words, subtle natural movements.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary motion, camera, or atmosphere while preserving the user's specified elements. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same source/end frames and generation parameters and only prompt text varies. Example: \"{gentle sway with soft birdsong|dramatic zoom with rolling thunder|slow pan with ambient music}\"."
1050
1050
  },
1051
1051
  "expandPrompt": {
1052
1052
  "type": "boolean",
@@ -1059,6 +1059,7 @@ exports.generationToolsManifest = {
1059
1059
  "videoModel": {
1060
1060
  "type": "string",
1061
1061
  "enum": [
1062
+ "ltx25",
1062
1063
  "ltx23",
1063
1064
  "wan22",
1064
1065
  "happyhorse-1.1-i2v",
@@ -1068,7 +1069,7 @@ exports.generationToolsManifest = {
1068
1069
  "minimax-h3-flf2v",
1069
1070
  "minimax-h3-flf2v-turbo"
1070
1071
  ],
1071
- "description": "Which video model to use. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step variant and Default Media Quality Pro uses the non-distilled dev variant; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, seedance2-fast, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo is available only as minimax-h3-t2v-turbo, minimax-h3-i2v-turbo, and minimax-h3-flf2v-turbo; there is no Turbo R2V."
1072
+ "description": "Which video model to use. \"ltx25\" (default): LTX 2.5 with native audio; Fast/HQ use the official distilled INT8 workflow and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2; per-clip duration 2-20s. \"ltx23\": LTX 2.3 rollback with its existing distilled/dev quality routing; per-clip duration 2-20s. \"wan22\": Fast 4-step, simple motion, no audio; per-clip duration capped at 10s. \"happyhorse-1.1-i2v\": HappyHorse 1.1 image-to-video from one source frame; 3-15s, 720p/1080p, native synchronized audio always on. \"happyhorse-1.1-r2v\": HappyHorse 1.1 reference-to-video with image-only loose references; use only when referenceImageIndices provide additional image references for the same clip. \"minimax-h3-i2v\": MiniMax H3 image-to-video from one first frame; 5.17-15.08s at a fixed 24 fps with jointly generated stereo audio, a 1344x768 pixel budget on a 32px grid. \"minimax-h3-flf2v\": MiniMax H3 first-and-last-frame interpolation; use it with frameRole=\"both\" plus endImageIndex/endImageIndices so the first image is the opening frame and the second is the closing frame. Do not set seedance2, seedance2-mini, seedance2-fast, or seedance2-5 here; use generate_video with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and @Image/@Video/@Audio role text for Seedance. HappyHorse accepts neither negativePrompt nor generateAudio. MiniMax H3 accepts no negativePrompt; set generateAudio=false only when the user asks for silent output, and the returned video has no audio track. MiniMax H3 Base and Turbo T2V/I2V/FLF2V prompts use exactly integrated_multimodal_description, overall_soundscape, then non_diegetic_music; I2V/FLF2V prepend the official alignment line. Standard uses 20 steps with res_multistep/simple; Turbo uses 4 steps with er_sde/simple. Turbo T2V/I2V/FLF2V use the selectors above; the dedicated Turbo R2V selector minimax-h3-r2v-turbo is intentionally routed through generate_video because it takes a loose multi-reference set rather than frame anchors."
1072
1073
  },
1073
1074
  "generateAudio": {
1074
1075
  "type": "boolean",
@@ -1076,11 +1077,11 @@ exports.generationToolsManifest = {
1076
1077
  },
1077
1078
  "negativePrompt": {
1078
1079
  "type": "string",
1079
- "description": "Advanced LTX/WAN only. Use this field only when the user explicitly asks to set a separate negative prompt. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt."
1080
+ "description": "Advanced LTX 2.5/LTX 2.3/WAN only. All standard LTX 2.5 image and first/last-frame workflows accept this separate negative prompt. Use this field only when the user explicitly asks to set one. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt."
1080
1081
  },
1081
1082
  "duration": {
1082
1083
  "type": "number",
1083
- "description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Per-model maximum: ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip."
1084
+ "description": "Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., \"make a 10 second video\"). Per-model maximum: ltx25 and ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip."
1084
1085
  },
1085
1086
  "targetResolution": {
1086
1087
  "type": "number",
@@ -1142,7 +1143,7 @@ exports.generationToolsManifest = {
1142
1143
  },
1143
1144
  "voicePersonaName": {
1144
1145
  "type": "string",
1145
- "description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23."
1146
+ "description": "ONLY when the user explicitly requests a registered/reference persona voice clip. Name of the persona whose voice clip to use as referenceAudioIdentity. Set this when the narrator/speaker is a different persona than the one shown in the video (e.g. \"David\" narrates a video of Aleyna), or to explicitly select a requested voice when multiple personas with voice clips are resolved. Do NOT set this for ordinary character dialogue, inferred voices, or personas without a voice clip — LTX 2.3 generates voice natively from the text prompt instead. Requires ltx23 because LTX 2.5 has no compatible ID-LoRA."
1146
1147
  }
1147
1148
  },
1148
1149
  "required": [
@@ -1186,13 +1187,13 @@ exports.generationToolsManifest = {
1186
1187
  "type": "function",
1187
1188
  "function": {
1188
1189
  "name": "video_to_video",
1189
- "description": "Transform an existing video using AI. Uses WAN 2.2 Animate (move/replace) with a reference image to animate a photo with the video's motion or swap the video's subject, LTX-2.3 V2V ControlNet (canny/pose/depth/detailer) for video-only transforms, or Seedance V2V when the user explicitly asks to transform, upscale, enhance, restyle, or remaster an uploaded video with Seedance. Requires an uploaded video file. Use when the user wants to animate a photo with video motion, replace subjects in a video, restyle an existing video, or enhance video quality.",
1190
+ "description": "Transform an existing video using WAN 2.2 Animate, LTX 2.5 V2V controls by default, LTX 2.3 as rollback, or Seedance V2V when explicitly requested. LTX 2.5 distilled supports canny/pose/depth/detailer/inpaint/outpaint; Dev + Speed LoRA supports canny/pose/depth/detailer. Requires an uploaded video.",
1190
1191
  "parameters": {
1191
1192
  "type": "object",
1192
1193
  "properties": {
1193
1194
  "prompt": {
1194
1195
  "type": "string",
1195
- "description": "Describe the TARGET appearance (not the transformation process). 2-4 present-tense sentences.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nFor LTX-2.3 canny/depth/pose modes, the source video preserves composition, depth, or motion. Spend prompt detail on style, atmosphere, lighting, surface texture, color palette, scale, and pacing.\n\nExamples by mode:\n- animate-move (DEFAULT — WAN 2.2 Animate Move: applies camera/motion from source video to reference image): \"Smooth cinematic camera movement following the subject through the scene.\"\n- animate-replace (WAN 2.2 Animate Replace: replaces the subject in the source video with the reference image): \"The person from the reference photo performing the actions from the video.\"\n- canny (LTX-2.3 — edge-detection restyle): \"Hand-drawn watercolor anime style with soft ink edges, muted teal and coral palette, rain mist, neon reflections, warm rim light, preserving original silhouettes and composition.\"\n- pose (LTX-2.3 — tracks skeleton, replace person): \"A glossy cartoon robot with exaggerated proportions, brushed metal texture, glowing cyan joints, energetic stage lighting, preserving the original dance timing and pose.\"\n- depth (LTX-2.3 — depth-map restyle): \"A misty alpine valley at golden hour, expansive scale, volumetric haze, cool blue shadows, warm rim light, cinematic depth, lingering continuous shot.\"\n- detailer (LTX-2.3 — enhance quality): DESCRIBE THE SOURCE, do not request changes. Append quality qualifiers only. E.g. \"The same scene, ultra-sharp and clean, crisp high-resolution detail, preserving all original content, composition, and color.\" Avoid words like \"enhanced textures\", \"restyled\", or any new subjects/objects — they cause drift.\n- seedance-v2v (BytePlus Dreamina Seedance V2V): \"Restyle the source clip in a watercolor look with soft ink edges, while preserving its motion and composition.\" Use natural prose; Seedance reads the reference video holistically rather than via control-net constraints, so describe target style/mood/dialogue rather than control strength.\n\nPresent tense. Positive phrasing. Concrete visual details.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the artistic treatment while keeping control mode and structural intent consistent. Example: \"transform to {watercolor with soft edges|oil painting with bold strokes|anime with clean lines} style\"."
1196
+ "description": "Describe the TARGET appearance, motion, dialogue, audio, and style in positive present-tense language. For LTX 2.5 (default) or LTX 2.3 rollback canny/depth/pose modes, the source preserves the selected structure or motion, so emphasize style, atmosphere, lighting, texture, color, scale, and pacing. Canny preserves edges; pose preserves skeletal motion; depth preserves 3D layout; detailer should describe the original content with quality qualifiers only. Distilled LTX 2.5 also supports inpaint and outpaint; Dev + Speed LoRA does not. For inpaint, describe only the regenerated region. For outpaint, describe the newly revealed area consistently with the source. For Seedance V2V, use natural prose and describe the target transformation holistically."
1196
1197
  },
1197
1198
  "expandPrompt": {
1198
1199
  "type": "boolean",
@@ -1217,11 +1218,12 @@ exports.generationToolsManifest = {
1217
1218
  },
1218
1219
  "negativePrompt": {
1219
1220
  "type": "string",
1220
- "description": "Non-Seedance only. Optional negative prompt for LTX/Wan video-to-video models. Do not set when controlMode is seedance-v2v or videoModel is seedance2/seedance2-mini/seedance2-fast/seedance2-5; rewrite user-provided Seedance avoid/ban/no-X requests as positive prompt instructions."
1221
+ "description": "Non-Seedance only. Optional negative prompt supported by every LTX 2.5 and LTX 2.3 video-to-video control/edit template, plus WAN. Do not set when controlMode is seedance-v2v or videoModel is seedance2/seedance2-mini/seedance2-fast/seedance2-5; rewrite user-provided Seedance avoid/ban/no-X requests as positive prompt instructions."
1221
1222
  },
1222
1223
  "videoModel": {
1223
1224
  "type": "string",
1224
1225
  "enum": [
1226
+ "ltx25-v2v",
1225
1227
  "ltx23-v2v",
1226
1228
  "wan22-animate",
1227
1229
  "seedance2",
@@ -1229,7 +1231,7 @@ exports.generationToolsManifest = {
1229
1231
  "seedance2-fast",
1230
1232
  "seedance2-5"
1231
1233
  ],
1232
- "description": "Model selector for this video-to-video request. Usually omit; controlMode chooses the non-Seedance model. For controlMode=\"seedance-v2v\", Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance V2V unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro or from 480p/720p resolution requests alone. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
1234
+ "description": "Model selector for this video-to-video request. Usually omit; controlMode chooses the non-Seedance model. \"ltx25-v2v\" is the default LTX 2.5 control path; \"ltx23-v2v\" remains the rollback selector. For controlMode=\"seedance-v2v\", Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance V2V unless the user explicitly asks for legacy Fast, use \"seedance2-fast\" when the user asks for Seedance Fast / seedance-fast, and use \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro or from 480p/720p resolution requests alone. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
1233
1235
  },
1234
1236
  "generateAudio": {
1235
1237
  "type": "boolean",
@@ -1438,25 +1440,29 @@ exports.generationToolsManifest = {
1438
1440
  "type": "function",
1439
1441
  "function": {
1440
1442
  "name": "sound_to_video",
1441
- "description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx23-ia2v; for lip-sync with a face image, use wan-s2v; if no image, use ltx23-a2v. If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.3 generates audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video instead. LONG AUDIO ON SEEDANCE: Seedance 2.0, Mini, and Fast cap each clip at 15s; Seedance 2.5 caps each clip at 30s, so prefer \"seedance2-5\" for 16-30s audio instead of splitting. When uploaded audio exceeds the selected Seedance model's per-clip cap, do NOT clamp and drop the rest — split the run into multiple sound_to_video calls in the same turn using 15s segments for seedance2/seedance2-mini/seedance2-fast or 30s segments for seedance2-5, then finish with a single stitch_video call referencing the resulting clip indices in order with audioIndex pointing at the same uploaded audio so the stitched output carries the full original soundtrack. LTX/WAN models accept up to 20s per clip, so single-call is fine for them.",
1443
+ "description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx25-ia2v by default (ltx23-ia2v is rollback); for lip-sync with a face image, use wan-s2v; if no image, use ltx25-a2v by default (ltx23-a2v is rollback). If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.5 is the default and generates audio natively; LTX 2.3 remains available as a rollback model and also generates audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video with videoModel=\"ltx23\" because LTX 2.5 has no compatible ID-LoRA. LONG AUDIO ON SEEDANCE: Seedance 2.0, Mini, and Fast cap each clip at 15s; Seedance 2.5 caps each clip at 30s, so prefer \"seedance2-5\" for 16-30s audio instead of splitting. When uploaded audio exceeds the selected Seedance model's per-clip cap, do NOT clamp and drop the rest — split the run into multiple sound_to_video calls in the same turn using 15s segments for seedance2/seedance2-mini/seedance2-fast or 30s segments for seedance2-5, then finish with a single stitch_video call referencing the resulting clip indices in order with audioIndex pointing at the same uploaded audio so the stitched output carries the full original soundtrack. LTX/WAN models accept up to 20s per clip, so single-call is fine for them.",
1442
1444
  "parameters": {
1443
1445
  "type": "object",
1444
1446
  "properties": {
1445
1447
  "prompt": {
1446
1448
  "type": "string",
1447
- "description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.3 image+audio mode, do not re-describe static details already visible in the reference image; focus on motion, action, camera, and how the image responds to the audio.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main visual beat + 1 simple camera move. Around 10s: 2-3 clear beats + 1 camera move. >10s: up to 4 beats in clear sequence. Let the audio define timing, but avoid stacking subject, camera, and environment motion in short clips.\n\nBLOCKING: Direct layout when it affects the shot: left/right placement, foreground/background, facing direction, and relative distance between subjects.\n\nLIP-SYNC: Shot framing, speaker's appearance and setting, physical performance synced to audio — gestures, expressions, jaw movement between phrases. Include acting beats.\n\nMUSIC VISUALIZATION: Visual style, environment, and how elements react to rhythm and energy.\n\nAUDIO-REACTIVE: Motion and visual changes that correspond to sounds in the track.\n\nLTX VOCABULARY: camera (tracking, dolly, pan, tilt, handheld, static frame), lighting/atmosphere (golden hour, neon glow, dramatic shadows, fog, rain, smoke, reflections), scale/pacing (expansive, epic, intimate, claustrophobic, slow motion, time-lapse, lingering shot, continuous shot), style/genre (film noir, painterly, cyberpunk, stop-motion, claymation, 2D/3D animation, hand-drawn, fantasy, thriller, experimental film).\n\nAVOID: Vague prompts, too many competing visual elements, abstract descriptions without visible behavior, rigid numeric constraints, readable text or logos. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the visual interpretation while keeping audio sync intent consistent. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same audio source/window, image source, model, duration, dimensions, and parameters and only prompt text varies. Example: \"{abstract neon visualization|nature scene with swaying trees|urban street with rain} synced to the beat\"."
1449
+ "description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.5 or LTX 2.3 image+audio mode, do not re-describe static details already visible in the reference image; focus on motion, action, camera, and how the image responds to the audio.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main visual beat + 1 simple camera move. Around 10s: 2-3 clear beats + 1 camera move. >10s: up to 4 beats in clear sequence. Let the audio define timing, but avoid stacking subject, camera, and environment motion in short clips.\n\nBLOCKING: Direct layout when it affects the shot: left/right placement, foreground/background, facing direction, and relative distance between subjects.\n\nLIP-SYNC: Shot framing, speaker's appearance and setting, physical performance synced to audio — gestures, expressions, jaw movement between phrases. Include acting beats.\n\nMUSIC VISUALIZATION: Visual style, environment, and how elements react to rhythm and energy.\n\nAUDIO-REACTIVE: Motion and visual changes that correspond to sounds in the track.\n\nLTX VOCABULARY: camera (tracking, dolly, pan, tilt, handheld, static frame), lighting/atmosphere (golden hour, neon glow, dramatic shadows, fog, rain, smoke, reflections), scale/pacing (expansive, epic, intimate, claustrophobic, slow motion, time-lapse, lingering shot, continuous shot), style/genre (film noir, painterly, cyberpunk, stop-motion, claymation, 2D/3D animation, hand-drawn, fantasy, thriller, experimental film).\n\nAVOID: Vague prompts, too many competing visual elements, abstract descriptions without visible behavior, rigid numeric constraints, readable text or logos. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the visual interpretation while keeping audio sync intent consistent. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same audio source/window, image source, model, duration, dimensions, and parameters and only prompt text varies. Example: \"{abstract neon visualization|nature scene with swaying trees|urban street with rain} synced to the beat\"."
1448
1450
  },
1449
1451
  "expandPrompt": {
1450
1452
  "type": "boolean",
1451
1453
  "description": "Seedance only. Whether to run the shared Seedance prompt shaper before dispatch. Defaults to true; set false only when the user explicitly asks to submit the compact prompt directly or not modify the prompt."
1452
1454
  },
1455
+ "negativePrompt": {
1456
+ "type": "string",
1457
+ "description": "Advanced LTX 2.5/LTX 2.3/WAN only. The LTX A2V and IA2V workflows accept this separate negative prompt. Use it only when the user explicitly asks to set one. Do not set for Seedance."
1458
+ },
1453
1459
  "audioSourceIndex": {
1454
1460
  "type": "number",
1455
1461
  "description": "Index of the uploaded audio file to use (0-based, from uploaded files list). If only one audio file is uploaded, use 0. If no audio was uploaded but generate_music was used earlier, omit this — the tool will automatically find the generated audio."
1456
1462
  },
1457
1463
  "sourceImageIndex": {
1458
1464
  "type": "number",
1459
- "description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.3 A2V), this is optional — omit it to generate video purely from text + audio."
1465
+ "description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.5 or LTX 2.3 A2V), this is optional — omit it to generate video purely from text + audio."
1460
1466
  },
1461
1467
  "audioStart": {
1462
1468
  "type": "number",
@@ -1465,7 +1471,7 @@ exports.generationToolsManifest = {
1465
1471
  },
1466
1472
  "duration": {
1467
1473
  "type": "number",
1468
- "description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.3 and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30. For music videos, use the MAXIMUM duration the selected model allows (20 for LTX/WAN, 30 for \"seedance2-5\") since the audio is always longer than the video limit. Use when the user explicitly requests a specific length.",
1474
+ "description": "Video duration in seconds. Default: 5. Range: 2-30; the usable window is per-model and the host clamps to it. LTX 2.5, LTX 2.3, and WAN accept 2-20, Seedance 2.0/Mini/Fast accept 4-15, and Seedance 2.5 accepts 4-30. For music videos, use the MAXIMUM duration the selected model allows (20 for LTX/WAN, 30 for \"seedance2-5\") since the audio is always longer than the video limit. Use when the user explicitly requests a specific length.",
1469
1475
  "minimum": 2,
1470
1476
  "maximum": 30
1471
1477
  },
@@ -1477,10 +1483,12 @@ exports.generationToolsManifest = {
1477
1483
  "seedance2-mini",
1478
1484
  "seedance2-fast",
1479
1485
  "seedance2-5",
1486
+ "ltx25-ia2v",
1487
+ "ltx25-a2v",
1480
1488
  "ltx23-ia2v",
1481
1489
  "ltx23-a2v"
1482
1490
  ],
1483
- "description": "Video model. \"ltx23-ia2v\" (default when image available): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step variant and Default Media Quality Pro uses the non-distilled dev variant. \"ltx23-a2v\" (default when no image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. \"seedance2-fast\": legacy Seedance 2.0 Fast. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance unless the user explicitly says Seedance Fast, pick \"seedance2-fast\" when the user says Seedance Fast / seedance-fast, and pick \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
1491
+ "description": "Video model. \"ltx25-ia2v\" (default when image available) and \"ltx25-a2v\" (default without an image) use LTX 2.5; Fast/HQ use official distilled INT8 and Pro uses dev INT8 with the required official distilled Speed LoRA in stage 2. \"ltx23-ia2v\" and \"ltx23-a2v\" retain the LTX 2.3 rollback paths with their existing quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-mini\": Seedance 2.0 Mini for faster, lower-cost 720p iteration. \"seedance2-fast\": legacy Seedance 2.0 Fast. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster draft/lower-cost Seedance unless the user explicitly says Seedance Fast, pick \"seedance2-fast\" when the user says Seedance Fast / seedance-fast, and pick \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, and a much larger reference budget than Seedance 2.0: up to 30 images, 10 videos, and 10 audios, with no more than 30 reference media files in total. Seedance 2.5 covers text-to-video, image-to-video from a first frame, image-to-video with both a first and a last frame, and multimodal reference-to-video (including video editing and video extension). Pick \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s natively instead of being split and stitched), or wants a first-and-last-frame Seedance transition. It does not replace \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance 2.5 is premium/subscriber-gated like the rest of the Seedance family."
1484
1492
  },
1485
1493
  "generateAudio": {
1486
1494
  "type": "boolean",
@@ -1533,13 +1541,14 @@ exports.generationToolsManifest = {
1533
1541
  "type": "string",
1534
1542
  "enum": [
1535
1543
  "auto",
1544
+ "ltx25",
1536
1545
  "ltx23",
1537
1546
  "seedance2",
1538
1547
  "seedance2-mini",
1539
1548
  "seedance2-fast",
1540
1549
  "seedance2-5"
1541
1550
  ],
1542
- "description": "Which model to use for the new segment. Default: \"auto\" — detect from the base video's producer (Seedance base → the same Seedance model, otherwise LTX-2.3). Override only when the user explicitly requests a different model. \"seedance2-5\" is the Seedance 2.5 extension path: 480p/720p only, 4-30s of new footage per call at 24 fps with native audio."
1551
+ "description": "Which model to use for the new segment. Default: \"auto\" — preserve Seedance for a Seedance base and otherwise use LTX 2.5. Use ltx23 only for explicit rollback. \"seedance2-5\" supports 480p/720p and 4-30s of new footage at 24 fps."
1543
1552
  },
1544
1553
  "keepOriginalAudio": {
1545
1554
  "type": "boolean",
@@ -1594,6 +1603,7 @@ exports.generationToolsManifest = {
1594
1603
  "type": "string",
1595
1604
  "enum": [
1596
1605
  "auto",
1606
+ "ltx25",
1597
1607
  "ltx23",
1598
1608
  "wan22",
1599
1609
  "seedance2",
@@ -1601,7 +1611,7 @@ exports.generationToolsManifest = {
1601
1611
  "seedance2-fast",
1602
1612
  "seedance2-5"
1603
1613
  ],
1604
- "description": "Which model to use for the new segment. Default: \"auto\" — detect from the base video's producer (Seedance base → the same Seedance model, Wan base → Wan 2.2, otherwise LTX-2.3). Override only when the user explicitly requests a different model. \"seedance2-5\" is the Seedance 2.5 editing path: 480p/720p only, 4-30s replacement windows at 24 fps with native audio."
1614
+ "description": "Which model to use for the new segment. Default: \"auto\" — preserve Seedance or WAN for matching base clips and otherwise use LTX 2.5. Use ltx23 only for explicit rollback. \"seedance2-5\" supports 480p/720p and 4-30s replacement windows at 24 fps."
1605
1615
  },
1606
1616
  "keepOriginalAudio": {
1607
1617
  "type": "boolean",