@sogni-ai/sogni-intelligence-client 3.33.0 → 4.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/dist/client/SogniClientWrapper.d.ts.map +1 -1
  2. package/dist/client/SogniClientWrapper.js +7 -4
  3. package/dist/client/SogniClientWrapper.js.map +1 -1
  4. package/dist/contracts/data/promptContracts.d.ts.map +1 -1
  5. package/dist/contracts/data/promptContracts.js +27 -16
  6. package/dist/contracts/data/promptContracts.js.map +1 -1
  7. package/dist/contracts/hostedToolValidation.d.ts.map +1 -1
  8. package/dist/contracts/hostedToolValidation.js +4 -0
  9. package/dist/contracts/hostedToolValidation.js.map +1 -1
  10. package/dist/contracts/imagePrompt.d.ts.map +1 -1
  11. package/dist/contracts/imagePrompt.js +2 -0
  12. package/dist/contracts/imagePrompt.js.map +1 -1
  13. package/dist/media/videoSettings.d.ts +17 -6
  14. package/dist/media/videoSettings.d.ts.map +1 -1
  15. package/dist/media/videoSettings.js +124 -18
  16. package/dist/media/videoSettings.js.map +1 -1
  17. package/dist/openai-tools/_manifests.generated.d.ts.map +1 -1
  18. package/dist/openai-tools/_manifests.generated.js +48 -25
  19. package/dist/openai-tools/_manifests.generated.js.map +1 -1
  20. package/dist/openai-tools/generation-tools.json +48 -25
  21. package/dist/public-skill-runtime/index.js +2 -2
  22. package/dist/public-skill-runtime/index.js.map +1 -1
  23. package/dist/schemas/tools/animate_photo.schema.json +5 -11
  24. package/dist/schemas/tools/generate_video.schema.json +16 -11
  25. package/dist/schemas/tools/sound_to_video.schema.json +13 -1
  26. package/dist/schemas/tools/video_to_video.schema.json +14 -2
  27. package/dist/tools/definitions/animate-photo/definition.d.ts.map +1 -1
  28. package/dist/tools/definitions/animate-photo/definition.js +6 -7
  29. package/dist/tools/definitions/animate-photo/definition.js.map +1 -1
  30. package/dist/tools/definitions/generate-video/definition.d.ts.map +1 -1
  31. package/dist/tools/definitions/generate-video/definition.js +16 -7
  32. package/dist/tools/definitions/generate-video/definition.js.map +1 -1
  33. package/dist/tools/definitions/sound-to-video/definition.d.ts.map +1 -1
  34. package/dist/tools/definitions/sound-to-video/definition.js +13 -1
  35. package/dist/tools/definitions/sound-to-video/definition.js.map +1 -1
  36. package/dist/tools/definitions/video-to-video/definition.d.ts.map +1 -1
  37. package/dist/tools/definitions/video-to-video/definition.js +14 -2
  38. package/dist/tools/definitions/video-to-video/definition.js.map +1 -1
  39. package/dist/tools/shared/modelRegistry.d.ts.map +1 -1
  40. package/dist/tools/shared/modelRegistry.js +3 -0
  41. package/dist/tools/shared/modelRegistry.js.map +1 -1
  42. package/dist/types/index.d.ts +4 -2
  43. package/dist/types/index.d.ts.map +1 -1
  44. package/dist/types/index.js.map +1 -1
  45. package/dist/utils/videoModelIds.d.ts.map +1 -1
  46. package/dist/utils/videoModelIds.js +3 -0
  47. package/dist/utils/videoModelIds.js.map +1 -1
  48. package/dist-esm/client/SogniClientWrapper.js +7 -4
  49. package/dist-esm/client/SogniClientWrapper.js.map +1 -1
  50. package/dist-esm/contracts/data/promptContracts.js +27 -16
  51. package/dist-esm/contracts/data/promptContracts.js.map +1 -1
  52. package/dist-esm/contracts/hostedToolValidation.js +4 -0
  53. package/dist-esm/contracts/hostedToolValidation.js.map +1 -1
  54. package/dist-esm/contracts/imagePrompt.js +2 -0
  55. package/dist-esm/contracts/imagePrompt.js.map +1 -1
  56. package/dist-esm/media/videoSettings.js +120 -16
  57. package/dist-esm/media/videoSettings.js.map +1 -1
  58. package/dist-esm/openai-tools/_manifests.generated.js +48 -25
  59. package/dist-esm/openai-tools/_manifests.generated.js.map +1 -1
  60. package/dist-esm/openai-tools/generation-tools.json +48 -25
  61. package/dist-esm/public-skill-runtime/index.js +2 -2
  62. package/dist-esm/public-skill-runtime/index.js.map +1 -1
  63. package/dist-esm/schemas/tools/animate_photo.schema.json +5 -11
  64. package/dist-esm/schemas/tools/generate_video.schema.json +16 -11
  65. package/dist-esm/schemas/tools/sound_to_video.schema.json +13 -1
  66. package/dist-esm/schemas/tools/video_to_video.schema.json +14 -2
  67. package/dist-esm/tools/definitions/animate-photo/definition.js +7 -8
  68. package/dist-esm/tools/definitions/animate-photo/definition.js.map +1 -1
  69. package/dist-esm/tools/definitions/generate-video/definition.js +17 -8
  70. package/dist-esm/tools/definitions/generate-video/definition.js.map +1 -1
  71. package/dist-esm/tools/definitions/sound-to-video/definition.js +13 -1
  72. package/dist-esm/tools/definitions/sound-to-video/definition.js.map +1 -1
  73. package/dist-esm/tools/definitions/video-to-video/definition.js +14 -2
  74. package/dist-esm/tools/definitions/video-to-video/definition.js.map +1 -1
  75. package/dist-esm/tools/shared/modelRegistry.js +3 -0
  76. package/dist-esm/tools/shared/modelRegistry.js.map +1 -1
  77. package/dist-esm/types/index.js.map +1 -1
  78. package/dist-esm/utils/videoModelIds.js +3 -0
  79. package/dist-esm/utils/videoModelIds.js.map +1 -1
  80. package/package.json +4 -4
@@ -236,6 +236,7 @@
236
236
  "minimax-h3-t2v",
237
237
  "minimax-h3-t2v-turbo",
238
238
  "minimax-h3-fasth3-t2v-turbo",
239
+ "minimax-h3-fasth3-t2v-turbo-2stage",
239
240
  "happyhorse-1.1-t2v",
240
241
  "happyhorse-1.1-i2v",
241
242
  "happyhorse-1.1-r2v",
@@ -244,7 +245,7 @@
244
245
  "wan3.0-video",
245
246
  "wan3.0-spicy-video"
246
247
  ],
247
- "description": "\"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow. The Dev checkpoints are not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Video model. \"ltx23\": LTX 2.3 rollback with native audio. \"wan22\": quick simple motion without audio. \"minimax-h3-t2v\": standard 20-step MiniMax H3 text-to-video; \"minimax-h3-t2v-turbo\": the existing 4-step LightX2V Turbo text-to-video; \"minimax-h3-fasth3-t2v-turbo\": the separate FastVideo VSA four-step FastH3 engine, about 2x faster and fixed to Euler/simple. All use native audio, fixed 24fps, 5.17-15.08s, and a 768p-class 32px-grid canvas; use animate_photo for H3 image-conditioned modes. Base and Turbo T2V/I2V/FLF2V prompts use the exact ordered fields integrated_multimodal_description, overall_soundscape, and non_diegetic_music; I2V/FLF2V prepend the official alignment line. \"minimax-h3-r2v\": standard 20-step MiniMax H3 reference-to-video; \"minimax-h3-r2v-turbo\": the dedicated LightX2V 4-step Ref2VA Turbo workflow using Euler/simple and a 960x544 default. FastH3 has no R2V mode. Both R2V selectors accept up to 9 images, 3 videos, and 3 audios (12 files total); at least one visual reference (image or video) is required and audio alone is invalid. Select references with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and address them with the official <Subject N>/<Picture N>/<Video N>/<Audio N> semantics. Seedance quality is selected only by model: use \"seedance2-mini\" for Seedance 2.0 Mini or faster/lower-cost 720p iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, first-and-last-frame conditioning, and a much larger reference budget than the 2.0 family: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Choose \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s in one call instead of being split and stitched), or wants a first-and-last-frame Seedance transition. Keep \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. Seedance supports multimodal loose reference assets: images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. Alibaba HappyHorse 1.1 video models (third-party vendor — requires Premium Spark). Select by mode: \"happyhorse-1.1-t2v\" for text-to-video, \"happyhorse-1.1-i2v\" for image-to-video from one first-frame image, and \"happyhorse-1.1-r2v\" for reference-to-video with up to 9 reference images. Resolutions 720P and 1080P; duration 3-15 seconds at 24 fps; native synchronized audio is always generated (do not set generateAudio or negativePrompt). Supported aspect ratios: 16:9, 9:16, 1:1, 4:3, 3:4, 4:5, 5:4, 9:21, 21:9. HappyHorse 1.1 takes image references only and renders a native synchronized audio track (always on; do not set generateAudio or a negative prompt). Pick the model by mode: happyhorse-1.1-t2v for text-to-video (no reference image), happyhorse-1.1-i2v for image-to-video from a single first frame, and happyhorse-1.1-r2v for reference-to-video with 1 to 9 reference images. For r2v, tag the images in the prompt as [Image 1]…[Image 9] and assign each a clear role. HappyHorse does not accept reference videos or reference audios. \"wan3.0-video\" is Alibaba Wan 3 and \"wan3.0-spicy-video\" is MuleRouter w3.0-video. Both render 2-30s at fixed 30 fps with optional native audio, provider prompt expansion, 480p/720p/1080p, adaptive/fixed ratios, first/last frames, and up to 10 image/5 video/5 audio references. Only Alibaba wan3.0-video accepts document/web context and watermark. Frame anchors and loose references are mutually exclusive. Do not send negativePrompt; video references are loose conditioning for a new result, not source-video editing or extension."
248
+ "description": "\"ltx25\" (default): LTX 2.5 with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow. The Dev checkpoints are not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Video model. \"ltx23\": LTX 2.3 rollback with native audio. \"wan22\": quick simple motion without audio. \"minimax-h3-t2v\": standard 20-step MiniMax H3 text-to-video; \"minimax-h3-t2v-turbo\": the existing 4-step LightX2V Turbo text-to-video; \"minimax-h3-fasth3-t2v-turbo\": the separate FastVideo VSA four-step FastH3 engine, about 2x faster and fixed to Euler/simple; \"minimax-h3-fasth3-t2v-turbo-2stage\": the two-stage FastH3 engine: FastH3 renders the canvas, then the worker enlarges it 2x and refines it, so the clip is delivered at twice the canvas width and height with the same length and audio. targetResolution picks the delivered class: 1080 renders a 544px short-edge canvas (960x544 is delivered at 1920x1088) for 10 Spark per second, 1440 or omitted renders the 768p canvas for 2K (1344x768 is delivered at 2688x1536) for 16 Spark per second, and 720 renders a 384px canvas (672x384 is delivered at 1344x768) for the regular FastH3 price of 4 Spark per second. The estimate prices every request. It takes the same inputs, durations and LoRAs as \"minimax-h3-fasth3-t2v-turbo\". Choose it when the user asks for 1080p, 1440p or 2K MiniMax H3 output, for two-stage output, or for the sharpest/best H3 quality; for ordinary 768p FastH3 output keep the regular FastH3 selector at targetResolution 768. All use native audio, fixed 24fps, 5.17-15.08s, and a 768p-class 32px-grid canvas; use animate_photo for H3 image-conditioned modes. Base and Turbo T2V/I2V/FLF2V prompts use the exact ordered fields integrated_multimodal_description, overall_soundscape, and non_diegetic_music; I2V/FLF2V prepend the official alignment line. \"minimax-h3-r2v\": standard 20-step MiniMax H3 reference-to-video; \"minimax-h3-r2v-turbo\": the dedicated LightX2V 4-step Ref2VA Turbo workflow using Euler/simple and a 960x544 default. FastH3 has no R2V mode. Both R2V selectors accept up to 9 images, 3 videos, and 3 audios (12 files total); at least one visual reference (image or video) is required and audio alone is invalid. Select references with referenceImageIndices/referenceVideoIndices/referenceAudioIndices and address them with the official <Subject N>/<Picture N>/<Video N>/<Audio N> semantics. Seedance quality is selected only by model: use \"seedance2-mini\" for Seedance 2.0 Mini or faster/lower-cost 720p iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or Mini. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p, 720p, and 1080p (4K is unsupported), 4-30s per clip at a fixed 24 fps, native audio, first-and-last-frame conditioning, and a much larger reference budget than the 2.0 family: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Choose \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s in one call instead of being split and stitched), or wants a first-and-last-frame Seedance transition. Keep \"seedance2\" for 4K requests; Seedance 2.5 supports up to 1080p. Seedance supports multimodal loose reference assets: images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices. Alibaba HappyHorse 1.1 video models (third-party vendor — requires Premium Spark). Select by mode: \"happyhorse-1.1-t2v\" for text-to-video, \"happyhorse-1.1-i2v\" for image-to-video from one first-frame image, and \"happyhorse-1.1-r2v\" for reference-to-video with up to 9 reference images. Resolutions 720P and 1080P; duration 3-15 seconds at 24 fps; native synchronized audio is always generated (do not set generateAudio or negativePrompt). Supported aspect ratios: 16:9, 9:16, 1:1, 4:3, 3:4, 4:5, 5:4, 9:21, 21:9. HappyHorse 1.1 takes image references only and renders a native synchronized audio track (always on; do not set generateAudio or a negative prompt). Pick the model by mode: happyhorse-1.1-t2v for text-to-video (no reference image), happyhorse-1.1-i2v for image-to-video from a single first frame, and happyhorse-1.1-r2v for reference-to-video with 1 to 9 reference images. For r2v, tag the images in the prompt as [Image 1]…[Image 9] and assign each a clear role. HappyHorse does not accept reference videos or reference audios. \"wan3.0-video\" is Alibaba Wan 3 and \"wan3.0-spicy-video\" is MuleRouter w3.0-video. Both render 2-30s at fixed 30 fps with optional native audio, provider prompt expansion, 480p/720p/1080p, adaptive/fixed ratios, first/last frames, and up to 10 image/5 video/5 audio references. Only Alibaba wan3.0-video accepts document/web context and watermark. Frame anchors and loose references are mutually exclusive. Do not send negativePrompt; video references are loose conditioning for a new result, not source-video editing or extension."
248
249
  },
249
250
  "generateAudio": {
250
251
  "type": "boolean",
@@ -281,15 +282,7 @@
281
282
  },
282
283
  "targetResolution": {
283
284
  "type": "number",
284
- "description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-mini\" vs \"seedance2-5\"). Seedance 2.0 full supports 4K; Seedance Mini and Seedance 2.5 support 480p/720p only, so never set 1080p or 4K for \"seedance2-5\". Wan 3 supports exactly 480p, 720p, and 1080p. HappyHorse supports only 720p and 1080p. Never set 4K for Wan 3 or HappyHorse. MiniMax H3 renders inside a 1344x768 pixel budget on a 32px grid, so use 768 for H3 and never 1080p or 4K. Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, Wan 3, HappyHorse, or MiniMax H3, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead. For MiniMax H3 2K requests keep targetResolution at 768 (or omit it) and set outputScale to 2 instead."
285
- },
286
- "outputScale": {
287
- "type": "integer",
288
- "enum": [
289
- 1,
290
- 2
291
- ],
292
- "description": "MiniMax H3 only. 2 delivers 2K output: the clip renders on the normal H3 canvas and is delivered at twice its width and height (1344x768 becomes 2688x1536) with the same length and audio, for +10 Spark per second (+6 at 480p). Set 2 only when the user asks for 2K, 1440p-class or extra-sharp H3 output; leave unset otherwise. Ignored for other models."
285
+ "description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-mini\" vs \"seedance2-5\"). Seedance 2.0 full supports 4K; Seedance Mini supports 480p/720p; Seedance 2.5 supports 480p/720p/1080p, so never set 4K for \"seedance2-5\". Wan 3 supports exactly 480p, 720p, and 1080p. HappyHorse supports only 720p and 1080p. Never set 4K for Wan 3 or HappyHorse. MiniMax H3 renders inside a 1344x768 pixel budget on a 32px grid, so use 768 for the regular H3 selectors and never 1080p or 4K. The two-stage H3 selector \"minimax-h3-fasth3-t2v-turbo-2stage\" delivers twice the canvas, so there targetResolution names the delivered short-edge class: 1080 (544px canvas short edge: 960x544 delivered at 1920x1088), 1440 for 2K (the 1344x768 canvas delivered at 2688x1536), or 720 (384px canvas: 672x384 delivered at 1344x768); omit it for 2K. Never set 4K for H3. Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, Wan 3, HappyHorse, or MiniMax H3, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
293
286
  },
294
287
  "numberOfVariations": {
295
288
  "type": "number",
@@ -313,7 +306,7 @@
313
306
  "type": "string",
314
307
  "minLength": 1
315
308
  },
316
- "description": "Ordered LoRA IDs to apply to a MiniMax H3 render. Use only when the user explicitly asks for a LoRA or for an effect one of these names describes. Stack up to 8 in one request; order matters because the adapters apply in sequence and do not commute. Keep this array positionally aligned with loraStrengths. The first render with an uncached LoRA takes longer to start while the worker downloads it.\n\nAccepted only when videoModel is one of \"minimax-h3-t2v\", \"minimax-h3-t2v-turbo\", \"minimax-h3-fasth3-t2v-turbo\", \"minimax-h3-r2v\", \"minimax-h3-r2v-turbo\". Every other video model on this tool loads no LoRAs and silently ignores these arrays, so set videoModel to an H3 mode in the same call when the user asks for one.\n\nFive LoRAs are published for MiniMax H3 today and the set differs by mode, so GET /v1/loras/comfy?modelId=<model> is authoritative for the mode in hand and carries exact ranges, maturity flags, and anything published since. h3-realism-people (fal) is a realism pass trained on live-action footage of people: it restores skin texture and pores, stray hairs, fabric weave and a fine sensor grain that the base model smooths away, and holds up in close-up. It is the only one gated on a trigger word — put r34l1sm near the FRONT of the prompt, or the render comes back as ordinary H3 with no error. h3-vbvr-video-reasoning is a prompt-adherence pass that holds the model to what was asked instead of improvising. h3-natural-face-speech (AdaptiveVision) makes people talking on camera look and sound more natural: cheeks, brows, jaw and lips move together as in real speech, and spoken English comes through clearer; use it for talking-head shots such as vlogs, podcasts, interviews and presenters. h3-better-motion (AdaptiveVision) gives people more natural, consistent body movement — weight shifts, strides, turns and gestures that follow through — for dance, sport, walking and other full-body shots. Both AdaptiveVision LoRAs work best with short, simple prompt sentences and are not validated on reference-to-video. h3-mystic-xxx-v4 is an uncensored adult fine-tune. Do not invent ids."
309
+ "description": "Ordered LoRA IDs to apply to a MiniMax H3 render. Use only when the user explicitly asks for a LoRA or for an effect one of these names describes. Stack up to 8 in one request; order matters because the adapters apply in sequence and do not commute. Keep this array positionally aligned with loraStrengths. The first render with an uncached LoRA takes longer to start while the worker downloads it.\n\nAccepted only when videoModel is one of \"minimax-h3-t2v\", \"minimax-h3-t2v-turbo\", \"minimax-h3-fasth3-t2v-turbo\", \"minimax-h3-fasth3-t2v-turbo-2stage\", \"minimax-h3-r2v\", \"minimax-h3-r2v-turbo\". Every other video model on this tool loads no LoRAs and silently ignores these arrays, so set videoModel to an H3 mode in the same call when the user asks for one.\n\nFive LoRAs are published for MiniMax H3 today and the set differs by mode, so GET /v1/loras/comfy?modelId=<model> is authoritative for the mode in hand and carries exact ranges, maturity flags, and anything published since. h3-realism-people (fal) is a realism pass trained on live-action footage of people: it restores skin texture and pores, stray hairs, fabric weave and a fine sensor grain that the base model smooths away, and holds up in close-up. It is the only one gated on a trigger word — put r34l1sm near the FRONT of the prompt, or the render comes back as ordinary H3 with no error. h3-vbvr-video-reasoning is a prompt-adherence pass that holds the model to what was asked instead of improvising. h3-natural-face-speech (AdaptiveVision) makes people talking on camera look and sound more natural: cheeks, brows, jaw and lips move together as in real speech, and spoken English comes through clearer; use it for talking-head shots such as vlogs, podcasts, interviews and presenters. h3-better-motion (AdaptiveVision) gives people more natural, consistent body movement — weight shifts, strides, turns and gestures that follow through — for dance, sport, walking and other full-body shots. Both AdaptiveVision LoRAs work best with short, simple prompt sentences and are not validated on reference-to-video. h3-mystic-xxx-v4 is an uncensored adult fine-tune. Do not invent ids."
317
310
  },
318
311
  "loraStrengths": {
319
312
  "type": "array",
@@ -323,6 +316,18 @@
323
316
  "type": "number"
324
317
  },
325
318
  "description": "Strength for each LoRA in loras, in the same order. Omitting the array applies 1.0 to every LoRA, which is NOT the catalog default and for h3-realism-people is already at the top of its band, so send explicit values. Video LoRAs are positive-only — unlike the bipolar Krea 2 image sliders, a negative value is not an inverse effect and 0 is off. h3-realism-people takes 0-2 and its catalog default is 0.8; 0.6-1 is the usable band. It also pulls the camera in as it climbs: at 1.5 and above the shot reliably recomposes and the grade darkens, which on an image-conditioned mode can crop the subject out of the frame the user supplied. Raise it above 1 only when the user asks for more, and prefer the default when they supplied a first or last frame. h3-vbvr-video-reasoning and h3-mystic-xxx-v4 both take 0-1 and do default to 1.0, with usable bands of 0.7-1 and 0.2-1. h3-natural-face-speech and h3-better-motion take 0-1.5 and default to 0.6; their usable band is 0.4-0.8."
319
+ },
320
+ "outputFormat": {
321
+ "type": "string",
322
+ "enum": [
323
+ "mp4",
324
+ "mov"
325
+ ],
326
+ "description": "Video container. Defaults to mp4. MOV is supported only by Seedance 2.5; choose it when the user requests MOV for editing."
327
+ },
328
+ "returnLastFrame": {
329
+ "type": "boolean",
330
+ "description": "Seedance 2.5 only. Set true to export a separate image of the final frame alongside the video. The result includes lastFrameUrl, which can be used as the first-frame image for a subsequent clip. Defaults to false; this does not extend the video automatically."
326
331
  }
327
332
  },
328
333
  "required": [
@@ -833,13 +838,15 @@
833
838
  "minimax-h3-i2v",
834
839
  "minimax-h3-i2v-turbo",
835
840
  "minimax-h3-fasth3-i2v-turbo",
841
+ "minimax-h3-fasth3-i2v-turbo-2stage",
836
842
  "minimax-h3-flf2v",
837
843
  "minimax-h3-flf2v-turbo",
838
844
  "minimax-h3-fasth3-flf2v-turbo",
845
+ "minimax-h3-fasth3-flf2v-turbo-2stage",
839
846
  "wan3.0-video",
840
847
  "wan3.0-spicy-video"
841
848
  ],
842
- "description": "\"ltx25\" (default): LTX 2.5 I2V or first/last-frame video with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow. The Dev checkpoints are not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Which video model to use. \"ltx23\": LTX 2.3 rollback with native audio. \"wan22\": quick simple motion without audio, up to 10s. \"minimax-h3-i2v\" is standard MiniMax H3 from one first frame; \"minimax-h3-i2v-turbo\" is the existing 4-step LightX2V Turbo engine; \"minimax-h3-fasth3-i2v-turbo\" is the separate FastVideo VSA four-step FastH3 engine, about 2x faster and fixed to Euler/simple. The matching FLF2V selectors provide standard, LightX2V Turbo, and FastH3 Turbo first/last-frame generation; FastH3 has no R2V mode; use frameRole=\"both\" and provide the end frame. H3 generates native audio at fixed 24fps for 5.17-15.08s and has no negative-prompt input. H3 Base and Turbo prompts use the exact three-field contract and the official mode-specific alignment line. Do not set Seedance here; use generate_video with Seedance references. \"wan3.0-video\" is Alibaba Wan 3 and \"wan3.0-spicy-video\" is MuleRouter w3.0-video. Both render 2-30s at fixed 30 fps with optional native audio, provider prompt expansion, 480p/720p/1080p, adaptive/fixed ratios, first/last frames, and up to 10 image/5 video/5 audio references. Only Alibaba wan3.0-video accepts document/web context and watermark. Frame anchors and loose references are mutually exclusive. Do not send negativePrompt; video references are loose conditioning for a new result, not source-video editing or extension."
849
+ "description": "\"ltx25\" (default): LTX 2.5 I2V or first/last-frame video with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow. The Dev checkpoints are not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Which video model to use. \"ltx23\": LTX 2.3 rollback with native audio. \"wan22\": quick simple motion without audio, up to 10s. \"minimax-h3-i2v\" is standard MiniMax H3 from one first frame; \"minimax-h3-i2v-turbo\" is the existing 4-step LightX2V Turbo engine; \"minimax-h3-fasth3-i2v-turbo\" is the separate FastVideo VSA four-step FastH3 engine, about 2x faster and fixed to Euler/simple. \"minimax-h3-fasth3-i2v-turbo-2stage\" (first frame) and \"minimax-h3-fasth3-flf2v-turbo-2stage\" (first and last frame) are the two-stage FastH3 engine: FastH3 renders the canvas, then the worker enlarges it 2x and refines it, so the clip is delivered at twice the canvas width and height with the same length and audio. targetResolution picks the delivered class: 1080 renders a 544px short-edge canvas (960x544 is delivered at 1920x1088) for 10 Spark per second, 1440 or omitted renders the 768p canvas for 2K (1344x768 is delivered at 2688x1536) for 16 Spark per second, and 720 renders a 384px canvas (672x384 is delivered at 1344x768) for the regular FastH3 price of 4 Spark per second. The estimate prices every request. They take the same inputs, durations and LoRAs as their FastH3 selectors. Choose them when the user asks for 1080p, 1440p or 2K MiniMax H3 output, for two-stage output, or for the sharpest/best H3 quality; for ordinary 768p FastH3 output keep the regular FastH3 selector at targetResolution 768. The matching FLF2V selectors provide standard, LightX2V Turbo, FastH3 Turbo, and two-stage FastH3 first/last-frame generation; FastH3 has no R2V mode; use frameRole=\"both\" and provide the end frame. H3 generates native audio at fixed 24fps for 5.17-15.08s and has no negative-prompt input. H3 Base and Turbo prompts use the exact three-field contract and the official mode-specific alignment line. Do not set Seedance here; use generate_video with Seedance references. \"wan3.0-video\" is Alibaba Wan 3 and \"wan3.0-spicy-video\" is MuleRouter w3.0-video. Both render 2-30s at fixed 30 fps with optional native audio, provider prompt expansion, 480p/720p/1080p, adaptive/fixed ratios, first/last frames, and up to 10 image/5 video/5 audio references. Only Alibaba wan3.0-video accepts document/web context and watermark. Frame anchors and loose references are mutually exclusive. Do not send negativePrompt; video references are loose conditioning for a new result, not source-video editing or extension."
843
850
  },
844
851
  "negativePrompt": {
845
852
  "type": "string",
@@ -871,15 +878,7 @@
871
878
  },
872
879
  "targetResolution": {
873
880
  "type": "number",
874
- "description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"1080p\" without exact pixels or an output orientation. This preserves the source image aspect ratio. Wan 3 supports 480p, 720p, and 1080p; HappyHorse supports only 720p and 1080p. Never set 4K for either. MiniMax H3 renders inside a 1344x768 pixel budget on a 32px grid, so use 768 for H3 and never 1080p or 4K. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\" or \"720p landscape\", use exact-pixel aspectRatio instead. For MiniMax H3 2K requests keep targetResolution at 768 (or omit it) and set outputScale to 2 instead."
875
- },
876
- "outputScale": {
877
- "type": "integer",
878
- "enum": [
879
- 1,
880
- 2
881
- ],
882
- "description": "MiniMax H3 only. 2 delivers 2K output: the clip renders on the normal H3 canvas and is delivered at twice its width and height (1344x768 becomes 2688x1536) with the same length and audio, for +10 Spark per second (+6 at 480p). Set 2 only when the user asks for 2K, 1440p-class or extra-sharp H3 output; leave unset otherwise. Ignored for other models."
881
+ "description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"1080p\" without exact pixels or an output orientation. This preserves the source image aspect ratio. Wan 3 supports 480p, 720p, and 1080p; HappyHorse supports only 720p and 1080p. Never set 4K for either. MiniMax H3 renders inside a 1344x768 pixel budget on a 32px grid, so use 768 for the regular H3 selectors and never 1080p or 4K. The two-stage H3 selectors \"minimax-h3-fasth3-i2v-turbo-2stage\" and \"minimax-h3-fasth3-flf2v-turbo-2stage\" deliver twice the canvas, so there targetResolution names the delivered short-edge class: 1080 (544px canvas short edge: 960x544 delivered at 1920x1088), 1440 for 2K (the 1344x768 canvas delivered at 2688x1536), or 720 (384px canvas: 672x384 delivered at 1344x768); omit it for 2K. The source aspect is kept: a portrait source at 1080 renders 544x960 and is delivered at 1088x1920. Never set 4K for H3. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\" or \"720p landscape\", use exact-pixel aspectRatio instead."
883
882
  },
884
883
  "sourceImageIndex": {
885
884
  "type": "number",
@@ -947,7 +946,7 @@
947
946
  "type": "string",
948
947
  "minLength": 1
949
948
  },
950
- "description": "Ordered LoRA IDs to apply to a MiniMax H3 render. Use only when the user explicitly asks for a LoRA or for an effect one of these names describes. Stack up to 8 in one request; order matters because the adapters apply in sequence and do not commute. Keep this array positionally aligned with loraStrengths. The first render with an uncached LoRA takes longer to start while the worker downloads it.\n\nAccepted only when videoModel is one of \"minimax-h3-i2v\", \"minimax-h3-i2v-turbo\", \"minimax-h3-fasth3-i2v-turbo\", \"minimax-h3-flf2v\", \"minimax-h3-flf2v-turbo\", \"minimax-h3-fasth3-flf2v-turbo\". Every other video model on this tool loads no LoRAs and silently ignores these arrays, so set videoModel to an H3 mode in the same call when the user asks for one.\n\nFive LoRAs are published for MiniMax H3 today and the set differs by mode, so GET /v1/loras/comfy?modelId=<model> is authoritative for the mode in hand and carries exact ranges, maturity flags, and anything published since. h3-realism-people (fal) is a realism pass trained on live-action footage of people: it restores skin texture and pores, stray hairs, fabric weave and a fine sensor grain that the base model smooths away, and holds up in close-up. It is the only one gated on a trigger word — put r34l1sm near the FRONT of the prompt, or the render comes back as ordinary H3 with no error. h3-vbvr-video-reasoning is a prompt-adherence pass that holds the model to what was asked instead of improvising. h3-natural-face-speech (AdaptiveVision) makes people talking on camera look and sound more natural: cheeks, brows, jaw and lips move together as in real speech, and spoken English comes through clearer; use it for talking-head shots such as vlogs, podcasts, interviews and presenters. h3-better-motion (AdaptiveVision) gives people more natural, consistent body movement — weight shifts, strides, turns and gestures that follow through — for dance, sport, walking and other full-body shots. Both AdaptiveVision LoRAs work best with short, simple prompt sentences and are not validated on reference-to-video. h3-mystic-xxx-v4 is an uncensored adult fine-tune. Do not invent ids."
949
+ "description": "Ordered LoRA IDs to apply to a MiniMax H3 render. Use only when the user explicitly asks for a LoRA or for an effect one of these names describes. Stack up to 8 in one request; order matters because the adapters apply in sequence and do not commute. Keep this array positionally aligned with loraStrengths. The first render with an uncached LoRA takes longer to start while the worker downloads it.\n\nAccepted only when videoModel is one of \"minimax-h3-i2v\", \"minimax-h3-i2v-turbo\", \"minimax-h3-fasth3-i2v-turbo\", \"minimax-h3-fasth3-i2v-turbo-2stage\", \"minimax-h3-flf2v\", \"minimax-h3-flf2v-turbo\", \"minimax-h3-fasth3-flf2v-turbo\", \"minimax-h3-fasth3-flf2v-turbo-2stage\". Every other video model on this tool loads no LoRAs and silently ignores these arrays, so set videoModel to an H3 mode in the same call when the user asks for one.\n\nFive LoRAs are published for MiniMax H3 today and the set differs by mode, so GET /v1/loras/comfy?modelId=<model> is authoritative for the mode in hand and carries exact ranges, maturity flags, and anything published since. h3-realism-people (fal) is a realism pass trained on live-action footage of people: it restores skin texture and pores, stray hairs, fabric weave and a fine sensor grain that the base model smooths away, and holds up in close-up. It is the only one gated on a trigger word — put r34l1sm near the FRONT of the prompt, or the render comes back as ordinary H3 with no error. h3-vbvr-video-reasoning is a prompt-adherence pass that holds the model to what was asked instead of improvising. h3-natural-face-speech (AdaptiveVision) makes people talking on camera look and sound more natural: cheeks, brows, jaw and lips move together as in real speech, and spoken English comes through clearer; use it for talking-head shots such as vlogs, podcasts, interviews and presenters. h3-better-motion (AdaptiveVision) gives people more natural, consistent body movement — weight shifts, strides, turns and gestures that follow through — for dance, sport, walking and other full-body shots. Both AdaptiveVision LoRAs work best with short, simple prompt sentences and are not validated on reference-to-video. h3-mystic-xxx-v4 is an uncensored adult fine-tune. Do not invent ids."
951
950
  },
952
951
  "loraStrengths": {
953
952
  "type": "array",
@@ -1045,7 +1044,7 @@
1045
1044
  "seedance2-mini",
1046
1045
  "seedance2-5"
1047
1046
  ],
1048
- "description": "Model selector for this video-to-video request. Usually omit: non-Seedance controls default to \"ltx25-v2v\"; use \"ltx23-v2v\" only for rollback. LTX 2.5 Fast, HQ, and Pro currently use the release-validated official Distilled workflow for canny, pose, depth, detailer, inpaint, and outpaint. Dev is not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. For controlMode=\"seedance-v2v\", Seedance quality is selected only by model: use \"seedance2-mini\" for faster/lower-cost drafts and use \"seedance2\" for full-quality Seedance or 1080p/4K. \"seedance2-5\" supports 480p/720p, 4-30s at 24 fps, native audio, and first/last-frame conditioning; keep \"seedance2\" for 1080p/4K."
1047
+ "description": "Model selector for this video-to-video request. Usually omit: non-Seedance controls default to \"ltx25-v2v\"; use \"ltx23-v2v\" only for rollback. LTX 2.5 Fast, HQ, and Pro currently use the release-validated official Distilled workflow for canny, pose, depth, detailer, inpaint, and outpaint. Dev is not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. For controlMode=\"seedance-v2v\", Seedance quality is selected only by model: use \"seedance2-mini\" for faster/lower-cost drafts and use \"seedance2\" for full-quality Seedance or 1080p/4K. \"seedance2-5\" supports 480p/720p/1080p, 4-30s at 24 fps, native audio, and first/last-frame conditioning; keep \"seedance2\" for 4K."
1049
1048
  },
1050
1049
  "generateAudio": {
1051
1050
  "type": "boolean",
@@ -1053,7 +1052,7 @@
1053
1052
  },
1054
1053
  "targetResolution": {
1055
1054
  "type": "number",
1056
- "description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V Mini, Fast, and Seedance 2.5 support 480p and 720p only, so never set 1080p or 4K for \"seedance2-5\". Preserve the source video shape instead of forcing landscape pixels."
1055
+ "description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V Mini and Fast support 480p and 720p only; Seedance 2.5 also supports 1080p, so never set 4K for \"seedance2-5\". Preserve the source video shape instead of forcing landscape pixels."
1057
1056
  },
1058
1057
  "sourceImageIndex": {
1059
1058
  "type": "number",
@@ -1097,6 +1096,18 @@
1097
1096
  "description": "Number of video variations to generate (1-16). Default: 1.",
1098
1097
  "minimum": 1,
1099
1098
  "maximum": 16
1099
+ },
1100
+ "outputFormat": {
1101
+ "type": "string",
1102
+ "enum": [
1103
+ "mp4",
1104
+ "mov"
1105
+ ],
1106
+ "description": "Video container. Defaults to mp4. MOV is supported only by Seedance 2.5; choose it when the user requests MOV for editing."
1107
+ },
1108
+ "returnLastFrame": {
1109
+ "type": "boolean",
1110
+ "description": "Seedance 2.5 only. Set true to export a separate image of the final frame alongside the video. The result includes lastFrameUrl, which can be used as the first-frame image for a subsequent clip. Defaults to false; this does not extend the video automatically."
1100
1111
  }
1101
1112
  },
1102
1113
  "required": [
@@ -1354,7 +1365,7 @@
1354
1365
  "wan3.0-video",
1355
1366
  "wan3.0-spicy-video"
1356
1367
  ],
1357
- "description": "\"ltx25-ia2v\" (default with image) and \"ltx25-a2v\" (default without image): LTX 2.5 image+audio and audio-only modes; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflows. The Dev checkpoints are not publicly routed until upstream publishes and Sogni validates official ComfyUI Dev recipes. Video model. \"ltx23-ia2v\" (rollback with image): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (rollback without image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s. \"seedance2-mini\": Seedance 2.0 Mini, 720p cap, fastest/lower-cost Seedance option. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster/lower-cost drafts or explicit Mini requests, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p and 720p ONLY (it cannot render 1080p or 4K), 4-30s per clip at a fixed 24 fps, native audio, first-and-last-frame conditioning, and a much larger reference budget than the 2.0 family: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Choose \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s in one call instead of being split and stitched), or wants a first-and-last-frame Seedance transition. Keep \"seedance2\" for 1080p/4K requests, which Seedance 2.5 cannot satisfy. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"wan3.0-video\" is Alibaba Wan 3 and \"wan3.0-spicy-video\" is MuleRouter w3.0-video. Both render 2-30s at fixed 30 fps with optional native audio, provider prompt expansion, 480p/720p/1080p, adaptive/fixed ratios, first/last frames, and up to 10 image/5 video/5 audio references. Only Alibaba wan3.0-video accepts document/web context and watermark. Frame anchors and loose references are mutually exclusive. Do not send negativePrompt; video references are loose conditioning for a new result, not source-video editing or extension."
1368
+ "description": "\"ltx25-ia2v\" (default with image) and \"ltx25-a2v\" (default without image): LTX 2.5 image+audio and audio-only modes; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflows. The Dev checkpoints are not publicly routed until upstream publishes and Sogni validates official ComfyUI Dev recipes. Video model. \"ltx23-ia2v\" (rollback with image): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (rollback without image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s. \"seedance2-mini\": Seedance 2.0 Mini, 720p cap, fastest/lower-cost Seedance option. Seedance quality is selected only by this model value: pick \"seedance2-mini\" for faster/lower-cost drafts or explicit Mini requests, and pick \"seedance2\" for full-quality Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. \"seedance2-5\": Seedance 2.5, the newest Seedance generation — 480p, 720p, and 1080p (4K is unsupported), 4-30s per clip at a fixed 24 fps, native audio, first-and-last-frame conditioning, and a much larger reference budget than the 2.0 family: up to 30 images, 10 videos, and 10 audios, with up to 50 reference media files total, subject to those per-modality caps. Choose \"seedance2-5\" when the user asks for Seedance 2.5, wants a single continuous Seedance clip longer than 15s (2.5 renders up to 30s in one call instead of being split and stitched), or wants a first-and-last-frame Seedance transition. Keep \"seedance2\" for 4K requests; Seedance 2.5 supports up to 1080p. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present. \"wan3.0-video\" is Alibaba Wan 3 and \"wan3.0-spicy-video\" is MuleRouter w3.0-video. Both render 2-30s at fixed 30 fps with optional native audio, provider prompt expansion, 480p/720p/1080p, adaptive/fixed ratios, first/last frames, and up to 10 image/5 video/5 audio references. Only Alibaba wan3.0-video accepts document/web context and watermark. Frame anchors and loose references are mutually exclusive. Do not send negativePrompt; video references are loose conditioning for a new result, not source-video editing or extension."
1358
1369
  },
1359
1370
  "generateAudio": {
1360
1371
  "type": "boolean",
@@ -1373,6 +1384,18 @@
1373
1384
  "aspectRatio": {
1374
1385
  "type": "string",
1375
1386
  "description": "Do NOT set unless the user explicitly requests an aspect ratio, format, orientation, or exact pixel dimensions. When a reference/source image is used and the user did not ask to change its shape, omit this field so the handler preserves the selected source image's own ratio.\n\nFormats: \"16:9\", \"9:16\", \"4:5\", \"1:1\", \"4:3\", \"3:2\", \"21:9\", or exact pixels like \"1920x1080\".\n\nCRITICAL: When the user specifies exact pixel dimensions (e.g., \"1280x720\", \"1080x1920\", \"1920x1080\", \"3840x2160\") or an orientation-qualified named resolution (e.g., \"720p landscape\", \"720p portrait\"), use the exact pixel format, NOT a ratio like \"16:9\" or \"9:16\". Exact user-requested dimensions override the selected default media quality, including Pro/HQ defaults. A bare named video resolution like \"720p resolution\" is only a resolution tier/short-side request; do not turn it into landscape pixels and do not set aspectRatio unless the user also states landscape, portrait, vertical, horizontal, or exact pixels. If requested pixels are in bounds but not on the model's pixel step, still pass the user's exact pixel request; the handler snaps to the nearest supported size internally. Only use ratio format when the user says a generic format name without pixel dimensions.\n\nMappings (use ONLY when user does NOT specify pixel dimensions): landscape/widescreen/YouTube/cinematic → \"16:9\". portrait → \"9:16\". TikTok/Reels/IG Reels → \"1080x1920\". ultrawide/cinema scope → \"21:9\". Instagram post → \"4:5\". square → \"1:1\". standard/TV → \"4:3\". 720p landscape → \"1280x720\". 720p portrait → \"720x1280\". 1080p landscape → \"1920x1080\". 1080p portrait/HD portrait → \"1080x1920\". 4K landscape → \"3840x2160\". 4K portrait → \"2160x3840\". Never set for generic requests like \"make a video\"."
1387
+ },
1388
+ "outputFormat": {
1389
+ "type": "string",
1390
+ "enum": [
1391
+ "mp4",
1392
+ "mov"
1393
+ ],
1394
+ "description": "Video container. Defaults to mp4. MOV is supported only by Seedance 2.5; choose it when the user requests MOV for editing."
1395
+ },
1396
+ "returnLastFrame": {
1397
+ "type": "boolean",
1398
+ "description": "Seedance 2.5 only. Set true to export a separate image of the final frame alongside the video. The result includes lastFrameUrl, which can be used as the first-frame image for a subsequent clip. Defaults to false; this does not extend the video automatically."
1376
1399
  }
1377
1400
  },
1378
1401
  "required": [
@@ -771,7 +771,7 @@ exports.VIDEO_GENERATION_SKILL = {
771
771
  constraints: [
772
772
  'For My Personas video requests, default to image_editing first to produce a conditioned scene image before animation. Use direct video only when the user explicitly asks to animate an existing persona image/reference or no source image is available for a voice-only request.',
773
773
  'Wan 3.0 Enhanced uses exact Sogni model id wan3.0-spicy-video (MuleRouter provider id w3.0-video): 2-30 seconds at 30 fps, 480p/720p/1080p, native audio, prompt expansion, adaptive/fixed ratios, and up to 10 image/5 video/5 audio references. First/last-frame mode and loose-reference mode are mutually exclusive. It has no document/web context, watermark, negative prompt, source-video edit, or extend mode.',
774
- 'MiniMax H3 2K: every MiniMax H3 selector accepts outputScale=2, which renders on the normal H3 canvas and delivers twice the width and height (1344x768 becomes 2688x1536) with the same length and audio for +10 Spark per second (+6 at 480p). Set it only when the user asks for 2K, 1440p-class or extra-sharp H3 output, keep targetResolution at 768, and never set it for another model.',
774
+ 'MiniMax H3 two-stage: 1080p and 2K H3 delivery is its own selector, not an option. minimax-h3-fasth3-t2v-turbo-2stage on generate_video (animate_photo carries minimax-h3-fasth3-i2v-turbo-2stage and minimax-h3-fasth3-flf2v-turbo-2stage) renders the FastH3 canvas, then the worker enlarges it 2x and refines it, so the clip is delivered at twice the canvas with the same length and audio. targetResolution names the delivered class: 1080 renders a 544px short-edge canvas (960x544 becomes 1920x1088) for 10 Spark per second, 1440 or omitted renders the 768p canvas for 2K (1344x768 becomes 2688x1536) for 16 Spark per second, and 720 renders a 384px canvas (672x384 becomes 1344x768) for the regular FastH3 price of 4 Spark per second. Choose it when the user asks for 1080p, 1440p or 2K H3 output, for two-stage output, or for the sharpest/best H3 quality, and send the same prompt contract, durations and LoRAs as the matching FastH3 selector; keep ordinary 768p FastH3 output on the regular FastH3 selector at targetResolution 768.',
775
775
  ],
776
776
  };
777
777
  exports.VIDEO_EDITING_SKILL = {
@@ -1582,7 +1582,7 @@ exports.VIDEO_MODEL_REGISTRY = Object.freeze({
1582
1582
  defaultWidth: 1280,
1583
1583
  defaultHeight: 720,
1584
1584
  minDimension: 1,
1585
- maxDimension: 1280,
1585
+ maxDimension: 2206,
1586
1586
  dimensionMultiple: 1,
1587
1587
  fps: 24,
1588
1588
  frameStep: 1,