@sogni-ai/sogni-intelligence-client 3.23.4 → 3.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/contracts/data/promptContracts.d.ts.map +1 -1
- package/dist/contracts/data/promptContracts.js +23 -10
- package/dist/contracts/data/promptContracts.js.map +1 -1
- package/dist/contracts/toolPromptMarkers.d.ts +5 -5
- package/dist/contracts/toolPromptMarkers.d.ts.map +1 -1
- package/dist/contracts/toolPromptMarkers.js +5 -5
- package/dist/contracts/toolPromptMarkers.js.map +1 -1
- package/dist/media/enhancementProfiles.d.ts +1 -1
- package/dist/media/enhancementProfiles.d.ts.map +1 -1
- package/dist/media/enhancementProfiles.js +0 -1
- package/dist/media/enhancementProfiles.js.map +1 -1
- package/dist/media/vendorModelPremium.d.ts +1 -1
- package/dist/media/vendorModelPremium.d.ts.map +1 -1
- package/dist/media/vendorModelPremium.js +2 -1
- package/dist/media/vendorModelPremium.js.map +1 -1
- package/dist/media/videoSettings.d.ts +20 -1
- package/dist/media/videoSettings.d.ts.map +1 -1
- package/dist/media/videoSettings.js +36 -1
- package/dist/media/videoSettings.js.map +1 -1
- package/dist/openai-tools/_manifests.generated.d.ts.map +1 -1
- package/dist/openai-tools/_manifests.generated.js +292 -73
- package/dist/openai-tools/_manifests.generated.js.map +1 -1
- package/dist/openai-tools/generation-tools.json +292 -73
- package/dist/public-skill-runtime/index.d.ts +19 -2
- package/dist/public-skill-runtime/index.d.ts.map +1 -1
- package/dist/public-skill-runtime/index.js +55 -2
- package/dist/public-skill-runtime/index.js.map +1 -1
- package/dist/schemas/tools/animate_photo.schema.json +59 -19
- package/dist/schemas/tools/generate_video.schema.json +66 -18
- package/dist/schemas/tools/sound_to_video.schema.json +41 -12
- package/dist/schemas/tools/video_to_video.schema.json +79 -13
- package/dist/skills/asset_reference_management/modelRefRegistry.d.ts.map +1 -1
- package/dist/skills/asset_reference_management/modelRefRegistry.js +25 -0
- package/dist/skills/asset_reference_management/modelRefRegistry.js.map +1 -1
- package/dist/skills/asset_reference_management/types.d.ts +1 -1
- package/dist/skills/asset_reference_management/types.d.ts.map +1 -1
- package/dist/tools/definitions/animate-photo/definition.d.ts.map +1 -1
- package/dist/tools/definitions/animate-photo/definition.js +23 -5
- package/dist/tools/definitions/animate-photo/definition.js.map +1 -1
- package/dist/tools/definitions/generate-video/definition.d.ts.map +1 -1
- package/dist/tools/definitions/generate-video/definition.js +32 -9
- package/dist/tools/definitions/generate-video/definition.js.map +1 -1
- package/dist/tools/definitions/sound-to-video/definition.d.ts.map +1 -1
- package/dist/tools/definitions/sound-to-video/definition.js +30 -5
- package/dist/tools/definitions/sound-to-video/definition.js.map +1 -1
- package/dist/tools/definitions/video-to-video/definition.d.ts.map +1 -1
- package/dist/tools/definitions/video-to-video/definition.js +43 -13
- package/dist/tools/definitions/video-to-video/definition.js.map +1 -1
- package/dist/tools/index.d.ts +2 -0
- package/dist/tools/index.d.ts.map +1 -1
- package/dist/tools/index.js +9 -2
- package/dist/tools/index.js.map +1 -1
- package/dist/tools/shared/modelRegistry.d.ts.map +1 -1
- package/dist/tools/shared/modelRegistry.js +4 -0
- package/dist/tools/shared/modelRegistry.js.map +1 -1
- package/dist/tools/shared/wan3References.d.ts +36 -0
- package/dist/tools/shared/wan3References.d.ts.map +1 -0
- package/dist/tools/shared/wan3References.js +78 -0
- package/dist/tools/shared/wan3References.js.map +1 -0
- package/dist/utils/videoModelIds.d.ts +2 -1
- package/dist/utils/videoModelIds.d.ts.map +1 -1
- package/dist/utils/videoModelIds.js +12 -0
- package/dist/utils/videoModelIds.js.map +1 -1
- package/dist-esm/contracts/data/promptContracts.js +23 -10
- package/dist-esm/contracts/data/promptContracts.js.map +1 -1
- package/dist-esm/contracts/toolPromptMarkers.js +5 -5
- package/dist-esm/contracts/toolPromptMarkers.js.map +1 -1
- package/dist-esm/media/enhancementProfiles.js +0 -1
- package/dist-esm/media/enhancementProfiles.js.map +1 -1
- package/dist-esm/media/vendorModelPremium.js +2 -1
- package/dist-esm/media/vendorModelPremium.js.map +1 -1
- package/dist-esm/media/videoSettings.js +35 -0
- package/dist-esm/media/videoSettings.js.map +1 -1
- package/dist-esm/openai-tools/_manifests.generated.js +292 -73
- package/dist-esm/openai-tools/_manifests.generated.js.map +1 -1
- package/dist-esm/openai-tools/generation-tools.json +292 -73
- package/dist-esm/public-skill-runtime/index.js +52 -1
- package/dist-esm/public-skill-runtime/index.js.map +1 -1
- package/dist-esm/schemas/tools/animate_photo.schema.json +59 -19
- package/dist-esm/schemas/tools/generate_video.schema.json +66 -18
- package/dist-esm/schemas/tools/sound_to_video.schema.json +41 -12
- package/dist-esm/schemas/tools/video_to_video.schema.json +79 -13
- package/dist-esm/skills/asset_reference_management/modelRefRegistry.js +25 -0
- package/dist-esm/skills/asset_reference_management/modelRefRegistry.js.map +1 -1
- package/dist-esm/tools/definitions/animate-photo/definition.js +23 -5
- package/dist-esm/tools/definitions/animate-photo/definition.js.map +1 -1
- package/dist-esm/tools/definitions/generate-video/definition.js +32 -9
- package/dist-esm/tools/definitions/generate-video/definition.js.map +1 -1
- package/dist-esm/tools/definitions/sound-to-video/definition.js +30 -5
- package/dist-esm/tools/definitions/sound-to-video/definition.js.map +1 -1
- package/dist-esm/tools/definitions/video-to-video/definition.js +43 -13
- package/dist-esm/tools/definitions/video-to-video/definition.js.map +1 -1
- package/dist-esm/tools/index.js +1 -0
- package/dist-esm/tools/index.js.map +1 -1
- package/dist-esm/tools/shared/modelRegistry.js +4 -0
- package/dist-esm/tools/shared/modelRegistry.js.map +1 -1
- package/dist-esm/tools/shared/wan3References.js +71 -0
- package/dist-esm/tools/shared/wan3References.js.map +1 -0
- package/dist-esm/utils/videoModelIds.js +11 -0
- package/dist-esm/utils/videoModelIds.js.map +1 -1
- package/package.json +3 -3
|
@@ -1,23 +1,23 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
-
"$id": "https://schemas.sogni.ai/creative-agent/2026-
|
|
3
|
+
"$id": "https://schemas.sogni.ai/creative-agent/2026-07-18.1/tools/sound_to_video.schema.json",
|
|
4
4
|
"title": "sound_to_video arguments",
|
|
5
|
-
"schemaVersion": "2026-
|
|
6
|
-
"description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx25-ia2v by default (ltx23-ia2v is rollback); for lip-sync with a face image, use wan-s2v; if no image, use ltx25-a2v by default (ltx23-a2v is rollback). If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.5 and LTX 2.3 generate audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video with videoModel=\"ltx23\" because LTX 2.5 has no compatible ID-LoRA. LONG AUDIO ON SEEDANCE: Seedance 2.0 and Mini cap each clip at 15s; Seedance 2.5
|
|
5
|
+
"schemaVersion": "2026-07-18.1",
|
|
6
|
+
"description": "Generate video synchronized to audio. Use when the user has uploaded an audio file (mp3, wav, m4a, flac) and the audio is the primary sync target, especially uploaded-audio-only workflows. Also use after generate_music (\"turn that song into a video\", \"make a music video from that\"). Auto-detects generated audio from generate_music if no audio file is uploaded. Seedance animate_photo/generate_video can also attach uploaded audio as a loose @Audio reference when an image or video reference anchors the request; use this tool instead when the soundtrack itself should drive the video. If the user provides a reference image, use ltx25-ia2v by default (ltx23-ia2v is rollback); for lip-sync with a face image, use wan-s2v; if no image, use ltx25-a2v by default (ltx23-a2v is rollback). If the user wants dialogue/audio WITHOUT pre-existing audio, use animate_photo instead (LTX 2.5 and LTX 2.3 generate audio natively). Note: Persona voice clips from resolve_personas are NOT used by this tool — for persona voice identity in video, use animate_photo or generate_video with videoModel=\"ltx23\" because LTX 2.5 has no compatible ID-LoRA. LONG AUDIO ON SEEDANCE: Seedance 2.0 and Mini cap each clip at 15s; Seedance 2.5 renders up to 30s in one call, so prefer seedance2-5 for 16-30s audio instead of splitting. When the user uploads audio longer than the per-clip cap of the selected model and Seedance is selected (seedance2, seedance2-mini, or seedance2-5), do NOT clamp to 15s and drop the rest — split the run into multiple sound_to_video calls in the same turn (one per 15s segment, so a 20s audio becomes two clips: audioStart=0 duration=15, then audioStart=15 duration=5) and finish with a single stitch_video call referencing the resulting clip indices in order with audioIndex pointing at the same uploaded audio so the stitched output carries the full original soundtrack. LTX/WAN models accept up to 20s per clip, so single-call is fine for them. Use videoModel=\"wan3.0-video\" when the user explicitly requests Wan 3 audio-driven video.",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"properties": {
|
|
10
10
|
"prompt": {
|
|
11
11
|
"type": "string",
|
|
12
|
-
"description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.
|
|
12
|
+
"description": "Describe the video like a cinematographer. Let the audio define timing — use the prompt for visual interpretation. One flowing paragraph, present tense, specific natural language.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance or Wan 3, set expandPrompt=false.\n\nSTRUCTURE: shot/style and scale → subject → environment, lighting, color, texture, atmosphere → visual action synced to audio → camera movement. For LTX 2.3 image+audio mode, do not re-describe static details already visible in the reference image; focus on motion, action, camera, and how the image responds to the audio.\n\nMOTION PACING: Scale complexity to duration. <=6s: 1 main visual beat + 1 simple camera move. Around 10s: 2-3 clear beats + 1 camera move. >10s: up to 4 beats in clear sequence. Let the audio define timing, but avoid stacking subject, camera, and environment motion in short clips.\n\nBLOCKING: Direct layout when it affects the shot: left/right placement, foreground/background, facing direction, and relative distance between subjects.\n\nLIP-SYNC: Shot framing, speaker's appearance and setting, physical performance synced to audio — gestures, expressions, jaw movement between phrases. Include acting beats.\n\nMUSIC VISUALIZATION: Visual style, environment, and how elements react to rhythm and energy.\n\nAUDIO-REACTIVE: Motion and visual changes that correspond to sounds in the track.\n\nLTX VOCABULARY: camera (tracking, dolly, pan, tilt, handheld, static frame), lighting/atmosphere (golden hour, neon glow, dramatic shadows, fog, rain, smoke, reflections), scale/pacing (expansive, epic, intimate, claustrophobic, slow motion, time-lapse, lingering shot, continuous shot), style/genre (film noir, painterly, cyberpunk, stop-motion, claymation, 2D/3D animation, hand-drawn, fantasy, thriller, experimental film).\n\nAVOID: Vague prompts, too many competing visual elements, abstract descriptions without visible behavior, rigid numeric constraints, readable text or logos. QUOTING RULE: ONLY use double quotes for spoken dialogue. Never quote on-screen text, overlay text, titles, captions, signs, or any visual text — describe them without quotes.\n\nNON-SEEDANCE POSITIVE CONSTRAINTS: For ltx25-ia2v, ltx25-a2v, ltx23-ia2v, ltx23-a2v, and wan-s2v, prompt is a positive prompt. Translate user avoid/no/don't constraints into affirmative production constraints instead of copying negative phrasing. Preserve exact quoted visible text or dialogue when the user explicitly requests it; keep surrounding surfaces blank.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the visual interpretation while keeping audio sync intent consistent. This is one Sogni project with multiple jobs, so prefer it when all outputs share the same audio source/window, image source, model, duration, dimensions, and parameters and only prompt text varies. Example: \"{abstract neon visualization|nature scene with swaying trees|urban street with rain} synced to the beat\"."
|
|
13
13
|
},
|
|
14
14
|
"expandPrompt": {
|
|
15
15
|
"type": "boolean",
|
|
16
|
-
"description": "Seedance only. Whether to
|
|
16
|
+
"description": "Seedance and Wan 3 only. Whether to expand the prompt before dispatch. Defaults to true. For Wan 3, a successful Sogni expansion disables Alibaba prompt_extend to prevent a second rewrite; false disables both expansion layers so exact prompts remain exact."
|
|
17
17
|
},
|
|
18
18
|
"negativePrompt": {
|
|
19
19
|
"type": "string",
|
|
20
|
-
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. The LTX A2V and IA2V workflows accept this separate negative prompt. Use it only when the user explicitly asks to set one. Do not set for Seedance."
|
|
20
|
+
"description": "Advanced LTX 2.5/LTX 2.3/WAN only. The LTX A2V and IA2V workflows accept this separate negative prompt. Use it only when the user explicitly asks to set one. Do not set for Seedance.\n\nWan 3 has no negativePrompt request field; do not set this for wan3.0-video."
|
|
21
21
|
},
|
|
22
22
|
"audioSourceIndex": {
|
|
23
23
|
"type": "number",
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
},
|
|
26
26
|
"sourceImageIndex": {
|
|
27
27
|
"type": "number",
|
|
28
|
-
"description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.
|
|
28
|
+
"description": "Optional index of an uploaded image to use as the starting frame (0-based). Required for lip-sync models (WAN S2V). For audio-only-to-video models (LTX 2.3 A2V), this is optional — omit it to generate video purely from text + audio."
|
|
29
29
|
},
|
|
30
30
|
"audioStart": {
|
|
31
31
|
"type": "number",
|
|
@@ -34,10 +34,38 @@
|
|
|
34
34
|
},
|
|
35
35
|
"duration": {
|
|
36
36
|
"type": "number",
|
|
37
|
-
"description": "Video duration in seconds. Default: 5.
|
|
37
|
+
"description": "Video duration in seconds. Default: 5. Per-model range: LTX/WAN 2.2 = 2-20s; Wan 3 = 2-30s; Seedance 2.0 and Mini = 4-15s; Seedance 2.5 = 4-30s. For music videos, use the maximum duration the selected model allows because the audio is usually longer than the video limit. Use when the user explicitly requests a specific length.",
|
|
38
38
|
"minimum": 2,
|
|
39
39
|
"maximum": 30
|
|
40
40
|
},
|
|
41
|
+
"smartDuration": {
|
|
42
|
+
"type": "boolean",
|
|
43
|
+
"description": "Wan 3 only. Let the model choose 2-30 seconds. Do not also set duration. Sogni reserves 30 seconds and settles down to the provider-reported duration."
|
|
44
|
+
},
|
|
45
|
+
"ratio": {
|
|
46
|
+
"type": "string",
|
|
47
|
+
"enum": [
|
|
48
|
+
"adaptive",
|
|
49
|
+
"16:9",
|
|
50
|
+
"4:3",
|
|
51
|
+
"1:1",
|
|
52
|
+
"3:4",
|
|
53
|
+
"9:16"
|
|
54
|
+
],
|
|
55
|
+
"description": "Wan 3 only. Output ratio; \"adaptive\" derives it from the input."
|
|
56
|
+
},
|
|
57
|
+
"watermark": {
|
|
58
|
+
"type": "boolean",
|
|
59
|
+
"description": "Wan 3 only. Add Alibaba's visible watermark. Defaults to false."
|
|
60
|
+
},
|
|
61
|
+
"referenceFileUrl": {
|
|
62
|
+
"type": "string",
|
|
63
|
+
"description": "Wan 3 only. One public HTTPS document URL for additional audio-driven context (DOCX/DOC/XLSX/XLS/PPTX/PPT/PDF/TXT/KEY/PAGES/NUMBERS/Markdown, up to 100 MB; PDF/DOCX/DOC/PPTX/PPT/KEY/PAGES up to 50 pages). Mutually exclusive with referenceLinkUrl."
|
|
64
|
+
},
|
|
65
|
+
"referenceLinkUrl": {
|
|
66
|
+
"type": "string",
|
|
67
|
+
"description": "Wan 3 only. One public HTTPS webpage URL for additional audio-driven context. Mutually exclusive with referenceFileUrl."
|
|
68
|
+
},
|
|
41
69
|
"videoModel": {
|
|
42
70
|
"type": "string",
|
|
43
71
|
"enum": [
|
|
@@ -48,17 +76,18 @@
|
|
|
48
76
|
"ltx25-ia2v",
|
|
49
77
|
"ltx25-a2v",
|
|
50
78
|
"ltx23-ia2v",
|
|
51
|
-
"ltx23-a2v"
|
|
79
|
+
"ltx23-a2v",
|
|
80
|
+
"wan3.0-video"
|
|
52
81
|
],
|
|
53
|
-
"description": "
|
|
82
|
+
"description": "\"ltx25-ia2v\" (default with image) and \"ltx25-a2v\" (default without image) use the release-validated official LTX 2.5 Distilled/Turbo workflows for Fast, HQ, and Pro. Dev is not publicly routed until upstream publishes and Sogni validates official ComfyUI Dev recipes. \"ltx23-ia2v\" and \"ltx23-a2v\" remain rollback selectors with their existing quality-tier routing. \"wan-s2v\" is WAN 2.2 sound-to-video for lip-sync with a face image. Seedance quality is selected only by model: use \"seedance2-mini\" for faster/lower-cost 720p drafts, \"seedance2\" for full/non-fast Seedance or 1080p/4K, and \"seedance2-5\" for 480p/720p clips up to 30 seconds. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. Omit to auto-select based on whether an image is present. \"wan3.0-video\" is Alibaba Wan 3: one canonical premium-vendor model for text-to-video, first-frame and first+last-frame animation, loose multimodal references, audio-driven generation, and uploaded-video editing/extension. It renders 2-30s at fixed 30 fps with optional native audio, supports 480p/720p/1080p and 16:9/4:3/1:1/3:4/9:16, accepts up to 10 reference images, 5 reference videos, and 5 reference audios, and uses plain per-type prompt labels Image 1, Video 1, and Audio 1. Do not send negativePrompt. Use animate_photo for native first/last frames, generate_video for text or loose references, sound_to_video when audio is the primary driver, and video_to_video with controlMode=\"seedance-v2v\" for edits or extensions."
|
|
54
83
|
},
|
|
55
84
|
"generateAudio": {
|
|
56
85
|
"type": "boolean",
|
|
57
|
-
"description": "Whether the
|
|
86
|
+
"description": "Whether the returned video should include audio. Omit to include audio by default; set false when the user asks for silent output or no audio. The reference audio is still required and still drives generation even when the returned video has no audio track."
|
|
58
87
|
},
|
|
59
88
|
"numberOfVariations": {
|
|
60
89
|
"type": "number",
|
|
61
|
-
"description": "Number of video variations to generate (1-16). Default: 1.",
|
|
90
|
+
"description": "Number of video variations to generate (1-16). Use with one Dynamic Prompt branch when all variations share the same audio source/window, image source, model, duration, dimensions, and parameters and only prompt text varies. This creates one Sogni project with multiple jobs. Default: 1.",
|
|
62
91
|
"minimum": 1,
|
|
63
92
|
"maximum": 16
|
|
64
93
|
},
|
|
@@ -1,19 +1,19 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
-
"$id": "https://schemas.sogni.ai/creative-agent/2026-
|
|
3
|
+
"$id": "https://schemas.sogni.ai/creative-agent/2026-07-18.1/tools/video_to_video.schema.json",
|
|
4
4
|
"title": "video_to_video arguments",
|
|
5
|
-
"schemaVersion": "2026-
|
|
6
|
-
"description": "Transform an existing video using WAN 2.2 Animate, LTX 2.5 V2V controls by default, LTX 2.3 as rollback, or Seedance V2V when explicitly requested. LTX 2.5 Fast, HQ, and Pro use the release-validated official Distilled workflow for canny/pose/depth/detailer/inpaint/outpaint; Dev is not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Requires an uploaded video.",
|
|
5
|
+
"schemaVersion": "2026-07-18.1",
|
|
6
|
+
"description": "Transform an existing video using AI. Uses WAN 2.2 Animate (move/replace) with a reference image, LTX 2.5 V2V controls by default (canny/pose/depth/detailer plus distilled inpaint/outpaint), LTX 2.3 as rollback, or Seedance V2V when explicitly requested. LTX 2.5 Fast, HQ, and Pro use the release-validated official Distilled workflow for canny/pose/depth/detailer/inpaint/outpaint; Dev is not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Requires an uploaded video file. Use when the user wants to animate a photo with video motion, replace subjects, restyle footage, extend its canvas, regenerate a region, or enhance quality. Wan 3 source-video editing and continuation uses videoModel=\"wan3.0-video\" with controlMode=\"seedance-v2v\". Wan 3 source-video editing and continuation uses videoModel=\"wan3.0-video\" with controlMode=\"seedance-v2v\".",
|
|
7
7
|
"type": "object",
|
|
8
8
|
"additionalProperties": false,
|
|
9
9
|
"properties": {
|
|
10
10
|
"prompt": {
|
|
11
11
|
"type": "string",
|
|
12
|
-
"description": "Describe the TARGET appearance
|
|
12
|
+
"description": "Describe the TARGET appearance (not the transformation process). 2-4 present-tense sentences.\n\nLITERAL PROMPT OVERRIDE: If the user explicitly says not to modify the prompt, or to use it exactly/verbatim/as-is, copy the identified prompt text verbatim instead of applying these construction rules unless a hard requirement is missing. For Seedance or Wan 3, set expandPrompt=false.\n\nFor LTX 2.5 or 2.3 canny/depth/pose modes, the source video preserves composition, depth, or motion. Spend prompt detail on style, atmosphere, lighting, surface texture, color palette, scale, and pacing. LTX 2.5 Fast, HQ, and Pro use the release-validated official Distilled workflow for canny/pose/depth/detailer/inpaint/outpaint; Dev is not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe.\n\nExamples by mode:\n- animate-move (DEFAULT — WAN 2.2 Animate Move: applies camera/motion from source video to reference image): \"Smooth cinematic camera movement following the subject through the scene.\"\n- animate-replace (WAN 2.2 Animate Replace: replaces the subject in the source video with the reference image): \"The person from the reference photo performing the actions from the video.\"\n- canny (LTX 2.5 default; LTX 2.3 rollback — edge-detection restyle): \"Hand-drawn watercolor anime style with soft ink edges, muted teal and coral palette, rain mist, neon reflections, warm rim light, preserving original silhouettes and composition.\"\n- pose (LTX 2.5 or LTX 2.3 — tracks skeleton and transfers the reference-image subject): \"A glossy cartoon robot from the reference image performs the source video's motion, with brushed metal texture, glowing cyan joints, and energetic stage lighting.\" This mode requires a reference image as well as the source video.\n- depth (LTX 2.5 default; LTX 2.3 rollback — depth-map restyle): \"A misty alpine valley at golden hour, expansive scale, volumetric haze, cool blue shadows, warm rim light, cinematic depth, lingering continuous shot.\"\n- detailer (LTX 2.5 default; LTX 2.3 rollback — enhance quality): DESCRIBE THE SOURCE, do not request changes. Append quality qualifiers only. E.g. \"The same scene, ultra-sharp and clean, crisp high-resolution detail, preserving all original content, composition, and color.\" Avoid words like \"enhanced textures\", \"restyled\", or any new subjects/objects — they cause drift.\n- seedance-v2v (BytePlus Dreamina Seedance 2.0 V2V): \"Restyle the source clip in a watercolor look with soft ink edges, while preserving its motion and composition.\" Use natural prose; Seedance reads the reference video holistically rather than via control-net constraints, so describe target style/mood/dialogue rather than control strength.\n- outpaint (LTX 2.5 default; LTX 2.3 rollback — canvas extension): describe what fills the NEWLY REVEALED area around the original frame, consistent with the source scene. E.g. \"The same street scene continues seamlessly into the newly revealed space — more wet asphalt, parked cars, and glowing shopfronts, matching the original lighting and perspective.\" Set outpaintPosition (and optionally outpaintAspectRatio); no mask needed.\n- inpaint (LTX 2.5 default; LTX 2.3 rollback — masked region regeneration): describe ONLY what the inpainted region should become; the rest of the frame is preserved. E.g. \"A vintage red convertible parked at the curb, matching the scene's lighting and shadows.\" If the user supplied a mask, set maskImageIndex. If no mask was supplied, omit maskImageIndex so execution derives a mask from the source video and prompt.\n\nPresent tense. Positive phrasing. Concrete visual details.\n\nNON-SEEDANCE POSITIVE CONSTRAINTS: For LTX 2.5, LTX 2.3, and WAN 2.2 modes, prompt is a positive prompt. Translate user avoid/no/don't constraints into affirmative production constraints instead of copying negative phrasing. Preserve exact quoted visible text when the user explicitly requests it; keep surrounding surfaces blank.\n\nBATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary the artistic treatment while keeping control mode and structural intent consistent. Example: \"transform to {watercolor with soft edges|oil painting with bold strokes|anime with clean lines} style\"."
|
|
13
13
|
},
|
|
14
14
|
"expandPrompt": {
|
|
15
15
|
"type": "boolean",
|
|
16
|
-
"description": "Seedance only. Whether to
|
|
16
|
+
"description": "Seedance and Wan 3 only. Whether to expand the prompt before dispatch. Defaults to true. For Wan 3, a successful Sogni expansion disables Alibaba prompt_extend to prevent a second rewrite; false disables both expansion layers so exact prompts remain exact."
|
|
17
17
|
},
|
|
18
18
|
"videoSourceIndex": {
|
|
19
19
|
"type": "number",
|
|
@@ -28,13 +28,15 @@
|
|
|
28
28
|
"pose",
|
|
29
29
|
"depth",
|
|
30
30
|
"detailer",
|
|
31
|
+
"outpaint",
|
|
32
|
+
"inpaint",
|
|
31
33
|
"seedance-v2v"
|
|
32
34
|
],
|
|
33
|
-
"description": "How the source video and
|
|
35
|
+
"description": "How the source video and reference image interact. Pick by user intent:\n• \"animate-move\" (DEFAULT) — WAN 2.2 Animate Move. Applies camera movement and motion from the source video to the reference image, bringing a still photo to life. Requires sourceImageIndex.\n• \"animate-replace\" — WAN 2.2 Animate Replace. Replaces the subject in the source video with the person/character from the reference image, keeping the video's background and motion. Requires sourceImageIndex.\n• \"canny\" — LTX 2.5 (default) or 2.3 edge-detection control. Best for restyling while preserving exact composition and silhouettes. Video-only.\n• \"pose\" — LTX 2.5 (default) or 2.3 skeletal tracking. Best for replacing a person while keeping their motion. LTX 2.5 requires both the source video and sourceImageIndex for the subject appearance; LTX 2.3 keeps its existing optional-image rollback behavior.\n• \"depth\" — LTX 2.5 (default) or 2.3 depth-map control. Best for scenes with perspective, camera movement, or volumetric content. Video-only.\n• \"detailer\" — LTX 2.5 (default) or 2.3 quality enhancement. Describe the original scene with quality qualifiers and do not request content changes.\n• \"outpaint\" — distilled LTX 2.5 (default) or LTX 2.3 canvas extension. Set outpaintPosition and optionally outpaintAspectRatio; Pro/dev is not supported for this mode.\n• \"inpaint\" — LTX 2.5 by default (LTX 2.3 rollback) masked region regeneration. Regenerate or replace a specific region of the source video while preserving the rest (e.g. \"replace the billboard\", \"change what's on the table\"). If the user provides an uploaded mask image, set maskImageIndex to it. If no mask is provided, omit maskImageIndex; execution derives a mask from the source video and prompt before dispatch. The prompt describes only the target inpainted region. Video-only.\n• \"seedance-v2v\" — BytePlus Dreamina Seedance 2.0 video-to-video. Use only when the user explicitly asks for Seedance on the uploaded source video, such as a Seedance upscale, enhance, remaster, restyle, or transform. High-fidelity quality, native audio, time-coded scene control. Seedance V2V reads @Video1 holistically. Use it for restyling, motion transfer, extension, subject replacement, or scene transformation, and assign @Video1 a clear role such as source clip, camera movement, action timing, edit rhythm, or continuation anchor. Distinct from canny/depth/pose which use control-net constraints — Seedance treats the reference video holistically.\nCanny vs depth: canny preserves silhouettes and fine outlines — pick it for subject-led scenes and graphic restyles. Depth preserves 3D structure — pick it for scenes where the camera moves or spatial layout matters more than edge fidelity. Default: \"animate-move\".\n\nUse seedance-v2v with videoModel=\"wan3.0-video\" for Wan 3 source-video editing or continuation; describe Video 1 as the source in the prompt."
|
|
34
36
|
},
|
|
35
37
|
"negativePrompt": {
|
|
36
38
|
"type": "string",
|
|
37
|
-
"description": "
|
|
39
|
+
"description": "Advanced non-Seedance only. Use this field only when the user explicitly asks to set a separate negative prompt. For ordinary avoid/no/don't constraints on LTX 2.3 or WAN 2.2, translate them into affirmative production constraints inside prompt instead; do not move them here. Do not set when controlMode is seedance-v2v or videoModel is seedance2/seedance2-mini/seedance2-5.\n\nWan 3 has no negativePrompt request field; do not set this for wan3.0-video."
|
|
38
40
|
},
|
|
39
41
|
"videoModel": {
|
|
40
42
|
"type": "string",
|
|
@@ -44,28 +46,92 @@
|
|
|
44
46
|
"wan22-animate",
|
|
45
47
|
"seedance2",
|
|
46
48
|
"seedance2-mini",
|
|
47
|
-
"seedance2-5"
|
|
49
|
+
"seedance2-5",
|
|
50
|
+
"wan3.0-video"
|
|
48
51
|
],
|
|
49
|
-
"description": "Model selector for this video-to-video request. Usually omit
|
|
52
|
+
"description": "Model selector for this video-to-video request. Usually omit: LTX control modes choose \"ltx25-v2v\" by default, while \"ltx23-v2v\" remains an explicit rollback. LTX 2.5 Fast, HQ, and Pro currently use the release-validated official Distilled workflow for canny, depth, pose, detailer, inpaint, and outpaint. Dev is not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. Voice ID-LoRA, transition LoRA, and 10Eros remain LTX 2.3-only. For controlMode=\"seedance-v2v\", Seedance quality is selected only by model: use \"seedance2-mini\" for fast, lower-cost 720p Seedance V2V, and use \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro or from 480p/720p resolution requests alone. \"seedance2-5\" is 480p and 720p only, renders 4-30s at 24 fps, and supports native audio. \"wan3.0-video\" is Alibaba Wan 3: one canonical premium-vendor model for text-to-video, first-frame and first+last-frame animation, loose multimodal references, audio-driven generation, and uploaded-video editing/extension. It renders 2-30s at fixed 30 fps with optional native audio, supports 480p/720p/1080p and 16:9/4:3/1:1/3:4/9:16, accepts up to 10 reference images, 5 reference videos, and 5 reference audios, and uses plain per-type prompt labels Image 1, Video 1, and Audio 1. Do not send negativePrompt. Use animate_photo for native first/last frames, generate_video for text or loose references, sound_to_video when audio is the primary driver, and video_to_video with controlMode=\"seedance-v2v\" for edits or extensions."
|
|
50
53
|
},
|
|
51
54
|
"generateAudio": {
|
|
52
55
|
"type": "boolean",
|
|
53
|
-
"description": "Whether the
|
|
56
|
+
"description": "Whether the returned video should include generated or retained audio. Omit to include audio by default; set false when the user asks for silent output or no audio."
|
|
54
57
|
},
|
|
55
58
|
"targetResolution": {
|
|
56
59
|
"type": "number",
|
|
57
|
-
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V Mini and Seedance 2.5 support 480p and 720p only, so never set 1080p or 4K for \"seedance2-5\". Preserve the source video shape instead of forcing landscape pixels."
|
|
60
|
+
"description": "Seedance or Wan 3 V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V Mini, Fast, and Seedance 2.5 support 480p and 720p only, so never set 1080p or 4K for \"seedance2-5\". Wan 3 supports exactly 480p, 720p, and 1080p. Preserve the source video shape instead of forcing landscape pixels."
|
|
58
61
|
},
|
|
59
62
|
"sourceImageIndex": {
|
|
60
63
|
"type": "number",
|
|
61
|
-
"description": "
|
|
64
|
+
"description": "Index of a reference image (0-based). Required for \"animate-move\" and \"animate-replace\". Required for LTX 2.5 \"pose\" because that workflow needs both the source video and a still image that defines the subject appearance; optional for LTX 2.3 \"pose\" rollback. Ignored by \"canny\", \"depth\", \"detailer\", \"outpaint\", and \"inpaint\"."
|
|
65
|
+
},
|
|
66
|
+
"outpaintPosition": {
|
|
67
|
+
"type": "string",
|
|
68
|
+
"enum": [
|
|
69
|
+
"center",
|
|
70
|
+
"top",
|
|
71
|
+
"bottom",
|
|
72
|
+
"left",
|
|
73
|
+
"right"
|
|
74
|
+
],
|
|
75
|
+
"description": "controlMode=\"outpaint\" only. Where the ORIGINAL frame is anchored inside the expanded canvas, which determines the direction the canvas grows: \"left\" anchors the original on the left and adds new space on the right; \"right\" adds space on the left; \"top\" adds space below; \"bottom\" adds space above; \"center\" expands all sides evenly. Default: \"center\". Pick by the user's direction (\"extend to the right\" → \"left\"; \"make it wider\"/\"widescreen\" → \"center\")."
|
|
76
|
+
},
|
|
77
|
+
"outpaintAspectRatio": {
|
|
78
|
+
"type": "string",
|
|
79
|
+
"enum": [
|
|
80
|
+
"16:9",
|
|
81
|
+
"9:16",
|
|
82
|
+
"1:1",
|
|
83
|
+
"4:3",
|
|
84
|
+
"3:4",
|
|
85
|
+
"21:9"
|
|
86
|
+
],
|
|
87
|
+
"description": "controlMode=\"outpaint\" only. OPTIONAL target aspect ratio for the expanded canvas (e.g. \"16:9\" to make a vertical clip widescreen). The canvas only grows to reach this ratio — the original content is never cropped. Omit to expand moderately in the direction implied by outpaintPosition. Only set when the user names a target shape or orientation."
|
|
88
|
+
},
|
|
89
|
+
"maskImageIndex": {
|
|
90
|
+
"type": "number",
|
|
91
|
+
"description": "controlMode=\"inpaint\" only. Optional 0-based index of an uploaded mask IMAGE that marks the region to regenerate (white pixels = regenerate, black = preserve). Omit when the user did not provide a mask; execution will derive one from the source video and prompt. Ignored by every other controlMode."
|
|
62
92
|
},
|
|
63
93
|
"duration": {
|
|
64
94
|
"type": "number",
|
|
65
|
-
"description": "Output video duration in seconds.
|
|
95
|
+
"description": "Output video duration in seconds. Per-model range: WAN 2.2/LTX modes = 2-20s; Wan 3 = 2-30s subject to input-video plus output duration staying at or below 30s; Seedance 2.0 and Mini = 4-15s; Seedance 2.5 = 4-30s. If omitted, the tool matches the uploaded source video duration when available (capped to the selected model range); otherwise it falls back to 10s for WAN Animate Move/Replace and 5s for LTX/Seedance/Wan 3 modes. For long stitched/bulk WAN Animate Move/Replace work with no explicit per-clip length, prefer about 10s clips rather than 5s chunks. Only pass this when the user explicitly requests a different length.",
|
|
66
96
|
"minimum": 2,
|
|
67
97
|
"maximum": 30
|
|
68
98
|
},
|
|
99
|
+
"smartDuration": {
|
|
100
|
+
"type": "boolean",
|
|
101
|
+
"description": "Wan 3 only. Let Wan 3 choose 2-30 output seconds. Do not also set duration; input plus output must stay within the provider's 30-second limit. Sogni reserves 30 seconds and settles down to actual duration."
|
|
102
|
+
},
|
|
103
|
+
"wan3TaskType": {
|
|
104
|
+
"type": "string",
|
|
105
|
+
"enum": [
|
|
106
|
+
"edit",
|
|
107
|
+
"extend"
|
|
108
|
+
],
|
|
109
|
+
"description": "Wan 3 only. Use \"edit\" to transform the source or \"extend\" to continue it. Extension automatically uses ratio=\"adaptive\" and the prompt should explicitly describe continuation intent."
|
|
110
|
+
},
|
|
111
|
+
"ratio": {
|
|
112
|
+
"type": "string",
|
|
113
|
+
"enum": [
|
|
114
|
+
"adaptive",
|
|
115
|
+
"16:9",
|
|
116
|
+
"4:3",
|
|
117
|
+
"1:1",
|
|
118
|
+
"3:4",
|
|
119
|
+
"9:16"
|
|
120
|
+
],
|
|
121
|
+
"description": "Wan 3 only. Extension requires \"adaptive\"; edit defaults to adaptive."
|
|
122
|
+
},
|
|
123
|
+
"referenceFileUrl": {
|
|
124
|
+
"type": "string",
|
|
125
|
+
"description": "Wan 3 only. One public HTTPS document URL for additional edit/extension context (DOCX/DOC/XLSX/XLS/PPTX/PPT/PDF/TXT/KEY/PAGES/NUMBERS/Markdown, up to 100 MB; PDF/DOCX/DOC/PPTX/PPT/KEY/PAGES up to 50 pages). Mutually exclusive with referenceLinkUrl."
|
|
126
|
+
},
|
|
127
|
+
"referenceLinkUrl": {
|
|
128
|
+
"type": "string",
|
|
129
|
+
"description": "Wan 3 only. One public HTTPS webpage URL for additional edit/extension context. Mutually exclusive with referenceFileUrl."
|
|
130
|
+
},
|
|
131
|
+
"watermark": {
|
|
132
|
+
"type": "boolean",
|
|
133
|
+
"description": "Wan 3 only. Add Alibaba's visible watermark. Defaults to false."
|
|
134
|
+
},
|
|
69
135
|
"numberOfVariations": {
|
|
70
136
|
"type": "number",
|
|
71
137
|
"description": "Number of video variations to generate (1-16). Default: 1.",
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"modelRefRegistry.d.ts","sourceRoot":"","sources":["../../../src/skills/asset_reference_management/modelRefRegistry.ts"],"names":[],"mappings":"AAeA,OAAO,KAAK,EAAE,SAAS,EAAE,iBAAiB,EAAE,MAAM,YAAY,CAAC;AAK/D,MAAM,WAAW,cAAc;IAE7B,KAAK,EAAE,MAAM,CAAC;IAEd,IAAI,CAAC,EAAE,SAAS,CAAC;CAClB;AAED,MAAM,WAAW,cAAc;IAC7B,MAAM,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,SAAS,GAAG,MAAM,CAAC;IAC/C,KAAK,CAAC,KAAK,EAAE,MAAM,GAAG,cAAc,GAAG,IAAI,CAAC;IAE5C,SAAS,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,wBAAwB;IACvC,MAAM,EAAE,cAAc,CAAC;IACvB,QAAQ,EAAE,iBAAiB,GAAG,SAAS,CAAC;IACxC,SAAS,EAAE,OAAO,CAAC;CACpB;
|
|
1
|
+
{"version":3,"file":"modelRefRegistry.d.ts","sourceRoot":"","sources":["../../../src/skills/asset_reference_management/modelRefRegistry.ts"],"names":[],"mappings":"AAeA,OAAO,KAAK,EAAE,SAAS,EAAE,iBAAiB,EAAE,MAAM,YAAY,CAAC;AAK/D,MAAM,WAAW,cAAc;IAE7B,KAAK,EAAE,MAAM,CAAC;IAEd,IAAI,CAAC,EAAE,SAAS,CAAC;CAClB;AAED,MAAM,WAAW,cAAc;IAC7B,MAAM,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,SAAS,GAAG,MAAM,CAAC;IAC/C,KAAK,CAAC,KAAK,EAAE,MAAM,GAAG,cAAc,GAAG,IAAI,CAAC;IAE5C,SAAS,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,wBAAwB;IACvC,MAAM,EAAE,cAAc,CAAC;IACvB,QAAQ,EAAE,iBAAiB,GAAG,SAAS,CAAC;IACxC,SAAS,EAAE,OAAO,CAAC;CACpB;AA4KD,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,MAAM,GAAG,cAAc,CAEjE;AAED,wBAAgB,2BAA2B,CAAC,OAAO,EAAE,MAAM,GAAG,wBAAwB,CA4CrF;AAED,wBAAgB,wBAAwB,IAAI,aAAa,CAAC;IACxD,QAAQ,EAAE,iBAAiB,CAAC;IAC5B,MAAM,EAAE,cAAc,CAAC;CACxB,CAAC,CASD;AAED,wBAAgB,cAAc,CAAC,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,SAAS,GAAG,MAAM,CAEtF;AAED,wBAAgB,sBAAsB,IAAI,iBAAiB,EAAE,CAE5D"}
|
|
@@ -49,6 +49,27 @@ const HAPPYHORSE_FORMAT = {
|
|
|
49
49
|
},
|
|
50
50
|
scanRegex: /\[(?:Image|Video|Audio)\s+\d+\]/g,
|
|
51
51
|
};
|
|
52
|
+
const WAN3_FORMAT = {
|
|
53
|
+
format(index, type) {
|
|
54
|
+
if (type === 'video')
|
|
55
|
+
return `Video ${index}`;
|
|
56
|
+
if (type === 'audio')
|
|
57
|
+
return `Audio ${index}`;
|
|
58
|
+
return `Image ${index}`;
|
|
59
|
+
},
|
|
60
|
+
parse(token) {
|
|
61
|
+
const m = /^(Image|Video|Audio)\s+(\d+)$/.exec(token.trim());
|
|
62
|
+
if (!m)
|
|
63
|
+
return null;
|
|
64
|
+
const kind = m[1];
|
|
65
|
+
const idx = Number.parseInt(m[2], 10);
|
|
66
|
+
if (!Number.isFinite(idx) || idx < 1)
|
|
67
|
+
return null;
|
|
68
|
+
const t = kind === 'Video' ? 'video' : kind === 'Audio' ? 'audio' : 'image';
|
|
69
|
+
return { index: idx, type: t };
|
|
70
|
+
},
|
|
71
|
+
scanRegex: /\b(?:Image|Video|Audio)\s+\d+\b/g,
|
|
72
|
+
};
|
|
52
73
|
const GPT_IMAGE_2_FORMAT = {
|
|
53
74
|
format(index, type) {
|
|
54
75
|
if (type === 'video')
|
|
@@ -117,6 +138,7 @@ const MINIMAX_H3_FORMAT = {
|
|
|
117
138
|
const MODEL_REF_FORMATS = {
|
|
118
139
|
seedance: SEEDANCE_FORMAT,
|
|
119
140
|
happyhorse: HAPPYHORSE_FORMAT,
|
|
141
|
+
wan3: WAN3_FORMAT,
|
|
120
142
|
'gpt-image-2': GPT_IMAGE_2_FORMAT,
|
|
121
143
|
ltx23: CONTEXT_FORMAT,
|
|
122
144
|
ltx25: CONTEXT_FORMAT,
|
|
@@ -166,6 +188,9 @@ function getModelRefFormatResolution(modelId) {
|
|
|
166
188
|
if (videoFamily === 'happyhorse-1.1') {
|
|
167
189
|
return { format: HAPPYHORSE_FORMAT, model_id: 'happyhorse', fell_back: false };
|
|
168
190
|
}
|
|
191
|
+
if (videoFamily === 'wan3') {
|
|
192
|
+
return { format: WAN3_FORMAT, model_id: 'wan3', fell_back: false };
|
|
193
|
+
}
|
|
169
194
|
if (videoFamily === 'minimax-h3') {
|
|
170
195
|
return { format: MINIMAX_H3_FORMAT, model_id: 'minimax-h3', fell_back: false };
|
|
171
196
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"modelRefRegistry.js","sourceRoot":"","sources":["../../../src/skills/asset_reference_management/modelRefRegistry.ts"],"names":[],"mappings":";;
|
|
1
|
+
{"version":3,"file":"modelRefRegistry.js","sourceRoot":"","sources":["../../../src/skills/asset_reference_management/modelRefRegistry.ts"],"names":[],"mappings":";;AAkNA,8CAEC;AAED,kEA4CC;AAED,4DAYC;AAED,wCAEC;AAED,wDAEC;AAxQD,yEAAyE;AACzE,mEAAiF;AACjF,qFAA+F;AAsB/F,MAAM,eAAe,GAAmB;IACtC,MAAM,CAAC,KAAK,EAAE,IAAI;QAChB,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,SAAS,KAAK,EAAE,CAAC;QAC9C,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,SAAS,KAAK,EAAE,CAAC;QAC9C,OAAO,SAAS,KAAK,EAAE,CAAC;IAC1B,CAAC;IACD,KAAK,CAAC,KAAK;QACT,MAAM,CAAC,GAAG,6BAA6B,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC;QAC3D,IAAI,CAAC,CAAC;YAAE,OAAO,IAAI,CAAC;QACpB,MAAM,GAAG,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAE,EAAE,EAAE,CAAC,CAAC;QACvC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,CAAC;YAAE,OAAO,IAAI,CAAC;QAClD,MAAM,CAAC,GAAc,CAAC,CAAC,CAAC,CAAC,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC;QACvF,OAAO,EAAE,KAAK,EAAE,GAAG,EAAE,IAAI,EAAE,CAAC,EAAE,CAAC;IACjC,CAAC;IACD,SAAS,EAAE,4BAA4B;CACxC,CAAC;AASF,MAAM,iBAAiB,GAAmB;IACxC,MAAM,CAAC,KAAK,EAAE,IAAI;QAChB,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,UAAU,KAAK,GAAG,CAAC;QAChD,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,UAAU,KAAK,GAAG,CAAC;QAChD,OAAO,UAAU,KAAK,GAAG,CAAC;IAC5B,CAAC;IACD,KAAK,CAAC,KAAK;QACT,MAAM,CAAC,GAAG,mCAAmC,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC;QACjE,IAAI,CAAC,CAAC;YAAE,OAAO,IAAI,CAAC;QACpB,MAAM,IAAI,GAAG,CAAC,CAAC,CAAC,CAAE,CAAC;QACnB,MAAM,GAAG,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAE,EAAE,EAAE,CAAC,CAAC;QACvC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,CAAC;YAAE,OAAO,IAAI,CAAC;QAClD,MAAM,CAAC,GAAc,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC;QACvF,OAAO,EAAE,KAAK,EAAE,GAAG,EAAE,IAAI,EAAE,CAAC,EAAE,CAAC;IACjC,CAAC;IACD,SAAS,EAAE,kCAAkC;CAC9C,CAAC;AAMF,MAAM,WAAW,GAAmB;IAClC,MAAM,CAAC,KAAK,EAAE,IAAI;QAChB,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,SAAS,KAAK,EAAE,CAAC;QAC9C,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,SAAS,KAAK,EAAE,CAAC;QAC9C,OAAO,SAAS,KAAK,EAAE,CAAC;IAC1B,CAAC;IACD,KAAK,CAAC,KAAK;QACT,MAAM,CAAC,GAAG,+BAA+B,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC;QAC7D,IAAI,CAAC,CAAC;YAAE,OAAO,IAAI,CAAC;QACpB,MAAM,IAAI,GAAG,CAAC,CAAC,CAAC,CAAE,CAAC;QACnB,MAAM,GAAG,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAE,EAAE,EAAE,CAAC,CAAC;QACvC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,CAAC;YAAE,OAAO,IAAI,CAAC;QAClD,MAAM,CAAC,GAAc,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC;QACvF,OAAO,EAAE,KAAK,EAAE,GAAG,EAAE,IAAI,EAAE,CAAC,EAAE,CAAC;IACjC,CAAC;IACD,SAAS,EAAE,kCAAkC;CAC9C,CAAC;AAEF,MAAM,kBAAkB,GAAmB;IACzC,MAAM,CAAC,KAAK,EAAE,IAAI;QAChB,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,SAAS,KAAK,EAAE,CAAC;QAC9C,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,SAAS,KAAK,EAAE,CAAC;QAC9C,OAAO,SAAS,KAAK,EAAE,CAAC;IAC1B,CAAC;IACD,KAAK,CAAC,KAAK;QACT,MAAM,CAAC,GAAG,+BAA+B,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC;QAC7D,IAAI,CAAC,CAAC;YAAE,OAAO,IAAI,CAAC;QACpB,MAAM,IAAI,GAAG,CAAC,CAAC,CAAC,CAAE,CAAC;QACnB,MAAM,GAAG,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAE,EAAE,EAAE,CAAC,CAAC;QACvC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,CAAC;YAAE,OAAO,IAAI,CAAC;QAClD,MAAM,CAAC,GAAc,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC;QACvF,OAAO,EAAE,KAAK,EAAE,GAAG,EAAE,IAAI,EAAE,CAAC,EAAE,CAAC;IACjC,CAAC;IACD,SAAS,EAAE,iGAAiG;CAC7G,CAAC;AAMF,MAAM,cAAc,GAAmB;IACrC,MAAM,CAAC,KAAK,EAAE,IAAI;QAChB,MAAM,IAAI,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,KAAK,GAAG,CAAC,CAAC,CAAC;QACpC,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,iBAAiB,IAAI,EAAE,CAAC;QACrD,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,iBAAiB,IAAI,EAAE,CAAC;QACrD,OAAO,iBAAiB,IAAI,EAAE,CAAC;IACjC,CAAC;IACD,KAAK,CAAC,KAAK;QACT,MAAM,CAAC,GAAG,qCAAqC,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC;QACnE,IAAI,CAAC,CAAC;YAAE,OAAO,IAAI,CAAC;QACpB,MAAM,IAAI,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAE,EAAE,EAAE,CAAC,CAAC;QACxC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,IAAI,CAAC,IAAI,IAAI,GAAG,CAAC;YAAE,OAAO,IAAI,CAAC;QACpD,OAAO;YACL,KAAK,EAAE,IAAI,GAAG,CAAC;YACf,IAAI,EAAE,CAAC,CAAC,CAAC,CAAc;SACxB,CAAC;IACJ,CAAC;IACD,SAAS,EAAE,oCAAoC;CAChD,CAAC;AAWF,MAAM,iBAAiB,GAAmB;IACxC,MAAM,CAAC,KAAK,EAAE,IAAI;QAChB,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,UAAU,KAAK,GAAG,CAAC;QAChD,IAAI,IAAI,KAAK,OAAO;YAAE,OAAO,UAAU,KAAK,GAAG,CAAC;QAChD,OAAO,YAAY,KAAK,GAAG,CAAC;IAC9B,CAAC;IACD,KAAK,CAAC,KAAK;QACT,MAAM,CAAC,GAAG,mCAAmC,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC;QACjE,IAAI,CAAC,CAAC;YAAE,OAAO,IAAI,CAAC;QACpB,MAAM,IAAI,GAAG,CAAC,CAAC,CAAC,CAAE,CAAC;QACnB,MAAM,GAAG,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAE,EAAE,EAAE,CAAC,CAAC;QACvC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,GAAG,GAAG,CAAC;YAAE,OAAO,IAAI,CAAC;QAClD,MAAM,CAAC,GAAc,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,IAAI,KAAK,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC;QACvF,OAAO,EAAE,KAAK,EAAE,GAAG,EAAE,IAAI,EAAE,CAAC,EAAE,CAAC;IACjC,CAAC;IACD,SAAS,EAAE,kCAAkC;CAC9C,CAAC;AAEF,MAAM,iBAAiB,GAA8C;IACnE,QAAQ,EAAE,eAAe;IACzB,UAAU,EAAE,iBAAiB;IAC7B,IAAI,EAAE,WAAW;IACjB,aAAa,EAAE,kBAAkB;IACjC,KAAK,EAAE,cAAc;IACrB,KAAK,EAAE,cAAc;IACrB,GAAG,EAAE,cAAc;IACnB,iBAAiB,EAAE,cAAc;IACjC,oBAAoB,EAAE,cAAc;IACpC,IAAI,EAAE,kBAAkB;IACxB,YAAY,EAAE,iBAAiB;CAChC,CAAC;AAEF,SAAS,gBAAgB,CAAC,KAAa;IACrC,OAAO,IAAI,MAAM,CAAC,KAAK,CAAC,MAAM,EAAE,KAAK,CAAC,KAAK,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,CAAC,GAAG,KAAK,CAAC,KAAK,GAAG,CAAC,CAAC;AAC/F,CAAC;AAED,SAAS,mBAAmB,CAAC,OAAe;IAC1C,OAAO,OAAO,CAAC,IAAI,EAAE,CAAC,WAAW,EAAE,CAAC,OAAO,CAAC,UAAU,EAAE,GAAG,CAAC,CAAC,OAAO,CAAC,KAAK,EAAE,GAAG,CAAC,CAAC;AACnF,CAAC;AAED,SAAS,wBAAwB,CAAC,OAAe;IAC/C,MAAM,WAAW,GAAG,IAAA,oDAAiC,EAAC,OAAO,CAAC,CAAC;IAC/D,IAAI,WAAW,KAAK,OAAO;QAAE,OAAO,KAAK,CAAC;IAC1C,IAAI,WAAW,KAAK,OAAO;QAAE,OAAO,OAAO,CAAC;IAC5C,IAAI,WAAW,KAAK,OAAO,IAAI,WAAW,KAAK,MAAM;QAAE,OAAO,OAAO,CAAC;IACtE,MAAM,qBAAqB,GAAG,IAAA,kEAAsC,EAAC,OAAO,CAAC,CAAC;IAC9E,IAAI,qBAAqB,KAAK,iBAAiB;QAAE,OAAO,iBAAiB,CAAC;IAC1E,IAAI,qBAAqB,KAAK,oBAAoB;QAAE,OAAO,oBAAoB,CAAC;IAChF,OAAO,IAAI,CAAC;AACd,CAAC;AAMD,SAAgB,iBAAiB,CAAC,OAAe;IAC/C,OAAO,2BAA2B,CAAC,OAAO,CAAC,CAAC,MAAM,CAAC;AACrD,CAAC;AAED,SAAgB,2BAA2B,CAAC,OAAe;IACzD,MAAM,OAAO,GAAG,mBAAmB,CAAC,OAAO,CAAC,CAAC;IAC7C,IAAI,OAAO,IAAI,iBAAiB,EAAE,CAAC;QACjC,OAAO;YACL,MAAM,EAAE,iBAAiB,CAAC,OAA4B,CAAC;YACvD,QAAQ,EAAE,OAA4B;YACtC,SAAS,EAAE,KAAK;SACjB,CAAC;IACJ,CAAC;IACD,IAAI,IAAA,4CAAsB,EAAC,OAAO,CAAC,EAAE,CAAC;QACpC,OAAO,EAAE,MAAM,EAAE,eAAe,EAAE,QAAQ,EAAE,UAAU,EAAE,SAAS,EAAE,KAAK,EAAE,CAAC;IAC7E,CAAC;IACD,MAAM,WAAW,GAAG,IAAA,oDAAiC,EAAC,OAAO,CAAC,CAAC;IAC/D,IAAI,WAAW,KAAK,gBAAgB,EAAE,CAAC;QACrC,OAAO,EAAE,MAAM,EAAE,iBAAiB,EAAE,QAAQ,EAAE,YAAY,EAAE,SAAS,EAAE,KAAK,EAAE,CAAC;IACjF,CAAC;IACD,IAAI,WAAW,KAAK,MAAM,EAAE,CAAC;QAC3B,OAAO,EAAE,MAAM,EAAE,WAAW,EAAE,QAAQ,EAAE,MAAM,EAAE,SAAS,EAAE,KAAK,EAAE,CAAC;IACrE,CAAC;IACD,IAAI,WAAW,KAAK,YAAY,EAAE,CAAC;QACjC,OAAO,EAAE,MAAM,EAAE,iBAAiB,EAAE,QAAQ,EAAE,YAAY,EAAE,SAAS,EAAE,KAAK,EAAE,CAAC;IACjF,CAAC;IACD,MAAM,qBAAqB,GAAG,IAAA,kEAAsC,EAAC,OAAO,CAAC,CAAC;IAC9E,IAAI,qBAAqB,KAAK,aAAa,IAAI,qBAAqB,KAAK,MAAM,EAAE,CAAC;QAChF,OAAO;YACL,MAAM,EAAE,kBAAkB;YAC1B,QAAQ,EAAE,qBAAqB;YAC/B,SAAS,EAAE,KAAK;SACjB,CAAC;IACJ,CAAC;IACD,MAAM,iBAAiB,GAAG,wBAAwB,CAAC,OAAO,CAAC,CAAC;IAC5D,IAAI,iBAAiB,EAAE,CAAC;QACtB,OAAO;YACL,MAAM,EAAE,cAAc;YACtB,QAAQ,EAAE,iBAAiB;YAC3B,SAAS,EAAE,KAAK;SACjB,CAAC;IACJ,CAAC;IACD,OAAO,CAAC,IAAI,CAAC,iCAAiC,OAAO,8CAA8C,CAAC,CAAC;IACrG,OAAO;QACL,MAAM,EAAE,kBAAkB;QAC1B,QAAQ,EAAE,SAAS;QACnB,SAAS,EAAE,IAAI;KAChB,CAAC;AACJ,CAAC;AAED,SAAgB,wBAAwB;IAItC,OAAQ,MAAM,CAAC,OAAO,CAAC,iBAAiB,CAAgD;SACrF,GAAG,CAAC,CAAC,CAAC,QAAQ,EAAE,MAAM,CAAC,EAAE,EAAE,CAAC,CAAC;QAC5B,QAAQ;QACR,MAAM,EAAE;YACN,GAAG,MAAM;YACT,SAAS,EAAE,gBAAgB,CAAC,MAAM,CAAC,SAAS,CAAC;SAC9C;KACF,CAAC,CAAC,CAAC;AACR,CAAC;AAED,SAAgB,cAAc,CAAC,OAAe,EAAE,KAAa,EAAE,IAAe;IAC5E,OAAO,iBAAiB,CAAC,OAAO,CAAC,CAAC,MAAM,CAAC,KAAK,EAAE,IAAI,CAAC,CAAC;AACxD,CAAC;AAED,SAAgB,sBAAsB;IACpC,OAAO,MAAM,CAAC,IAAI,CAAC,iBAAiB,CAAwB,CAAC;AAC/D,CAAC"}
|
|
@@ -14,7 +14,7 @@ export interface AssetManifest {
|
|
|
14
14
|
next_index: number;
|
|
15
15
|
updated_at: string;
|
|
16
16
|
}
|
|
17
|
-
export type KnownAssetModelId = 'seedance' | 'happyhorse' | 'ltx23' | 'ltx25' | 'gpt-image-2' | 'wan' | 'qwen-image-edit' | 'krea-identity-edit' | 'flux' | 'minimax-h3';
|
|
17
|
+
export type KnownAssetModelId = 'seedance' | 'happyhorse' | 'wan3' | 'ltx23' | 'ltx25' | 'gpt-image-2' | 'wan' | 'qwen-image-edit' | 'krea-identity-edit' | 'flux' | 'minimax-h3';
|
|
18
18
|
export interface AssetReferenceValidationResult {
|
|
19
19
|
resolved: ReadonlyArray<{
|
|
20
20
|
token: string;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../src/skills/asset_reference_management/types.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../src/skills/asset_reference_management/types.ts"],"names":[],"mappings":"AAoBA,MAAM,MAAM,SAAS,GAAG,OAAO,GAAG,OAAO,GAAG,OAAO,CAAC;AAGpD,MAAM,WAAW,WAAW;IAE1B,QAAQ,EAAE,MAAM,CAAC;IAEjB,UAAU,EAAE,MAAM,CAAC;IAEnB,WAAW,CAAC,EAAE,MAAM,CAAC;IAErB,IAAI,EAAE,SAAS,CAAC;IAEhB,GAAG,CAAC,EAAE,MAAM,CAAC;IAEb,aAAa,CAAC,EAAE,SAAS,MAAM,EAAE,CAAC;IAElC,KAAK,CAAC,EAAE,SAAS,MAAM,EAAE,CAAC;IAE1B,QAAQ,CAAC,EAAE,QAAQ,CAAC,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC;CAC9C;AAGD,MAAM,WAAW,aAAa;IAC5B,MAAM,EAAE,SAAS,WAAW,EAAE,CAAC;IAE/B,UAAU,EAAE,MAAM,CAAC;IAEnB,UAAU,EAAE,MAAM,CAAC;CACpB;AAGD,MAAM,MAAM,iBAAiB,GACzB,UAAU,GACV,YAAY,GACZ,MAAM,GACN,OAAO,GACP,OAAO,GACP,aAAa,GACb,KAAK,GACL,iBAAiB,GACjB,oBAAoB,GACpB,MAAM,GACN,YAAY,CAAC;AAOjB,MAAM,WAAW,8BAA8B;IAE7C,QAAQ,EAAE,aAAa,CAAC;QACtB,KAAK,EAAE,MAAM,CAAC;QACd,QAAQ,EAAE,MAAM,CAAC;QACjB,UAAU,EAAE,MAAM,CAAC;KACpB,CAAC,CAAC;IAEH,QAAQ,EAAE,aAAa,CAAC;QACtB,KAAK,EAAE,MAAM,CAAC;QACd,MAAM,EAAE,eAAe,GAAG,mBAAmB,GAAG,oBAAoB,GAAG,wBAAwB,CAAC;QAChG,iBAAiB,CAAC,EAAE,MAAM,CAAC;KAC5B,CAAC,CAAC;IAOH,gBAAgB,EAAE,SAAS,MAAM,EAAE,CAAC;CACrC"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"definition.d.ts","sourceRoot":"","sources":["../../../../src/tools/definitions/animate-photo/definition.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;
|
|
1
|
+
{"version":3,"file":"definition.d.ts","sourceRoot":"","sources":["../../../../src/tools/definitions/animate-photo/definition.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AA8BlD,eAAO,MAAM,UAAU,EAAE,cAoLxB,CAAC"}
|
|
@@ -10,6 +10,7 @@ const H3_LORA_SELECTORS = [
|
|
|
10
10
|
"minimax-h3-flf2v",
|
|
11
11
|
"minimax-h3-flf2v-turbo",
|
|
12
12
|
];
|
|
13
|
+
const WAN3_VIDEO_MODEL_GUIDANCE = '"wan3.0-video" is Alibaba Wan 3: one canonical premium-vendor model for text-to-video, first-frame and first+last-frame animation, loose multimodal references, audio-driven generation, and uploaded-video editing/extension. It renders 2-30s at fixed 30 fps with optional native audio, supports 480p/720p/1080p and 16:9/4:3/1:1/3:4/9:16, accepts up to 10 reference images, 5 reference videos, and 5 reference audios, and uses plain per-type prompt labels Image 1, Video 1, and Audio 1. Do not send negativePrompt. Use animate_photo for native first/last frames, generate_video for text or loose references, sound_to_video when audio is the primary driver, and video_to_video with controlMode="seedance-v2v" for edits or extensions.';
|
|
13
14
|
exports.definition = {
|
|
14
15
|
type: "function",
|
|
15
16
|
function: {
|
|
@@ -72,25 +73,40 @@ BATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary
|
|
|
72
73
|
"minimax-h3-i2v-turbo",
|
|
73
74
|
"minimax-h3-flf2v",
|
|
74
75
|
"minimax-h3-flf2v-turbo",
|
|
76
|
+
"wan3.0-video",
|
|
75
77
|
],
|
|
76
78
|
description: '"ltx25" (default): LTX 2.5 I2V or first/last-frame video with native audio; Fast, HQ, and Pro currently use the release-validated official Distilled INT8 workflow. The Dev checkpoints are not publicly routed until upstream publishes and Sogni validates an official ComfyUI Dev recipe. ' +
|
|
77
|
-
'Which video model to use. "ltx23": LTX 2.3 rollback with native audio. "wan22": quick simple motion without audio, up to 10s. "minimax-h3-i2v" and "minimax-h3-i2v-turbo": standard and 4-step Turbo MiniMax H3 from one first frame. "minimax-h3-flf2v" and "minimax-h3-flf2v-turbo": standard and 4-step Turbo MiniMax H3 between required first and last frames; use frameRole="both" and provide the end frame. H3 generates native audio at fixed 24fps for 5.17-15.08s and has no negative-prompt input. H3 Base and Turbo prompts use the exact three-field contract and the official mode-specific alignment line. Do not set Seedance here; use generate_video with Seedance references.'
|
|
79
|
+
'Which video model to use. "ltx23": LTX 2.3 rollback with native audio. "wan22": quick simple motion without audio, up to 10s. "minimax-h3-i2v" and "minimax-h3-i2v-turbo": standard and 4-step Turbo MiniMax H3 from one first frame. "minimax-h3-flf2v" and "minimax-h3-flf2v-turbo": standard and 4-step Turbo MiniMax H3 between required first and last frames; use frameRole="both" and provide the end frame. H3 generates native audio at fixed 24fps for 5.17-15.08s and has no negative-prompt input. H3 Base and Turbo prompts use the exact three-field contract and the official mode-specific alignment line. Do not set Seedance here; use generate_video with Seedance references. ' +
|
|
80
|
+
WAN3_VIDEO_MODEL_GUIDANCE,
|
|
78
81
|
},
|
|
79
82
|
negativePrompt: {
|
|
80
83
|
type: "string",
|
|
81
|
-
description: "Advanced LTX/WAN only. Use this field only when the user explicitly asks to set a separate negative prompt. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt.",
|
|
84
|
+
description: "Advanced LTX/WAN only. Use this field only when the user explicitly asks to set a separate negative prompt. MiniMax H3 has no negative-prompt input; put requested exclusions in prompt.\n\nWan 3 has no negativePrompt request field; do not set this for wan3.0-video.",
|
|
82
85
|
},
|
|
83
86
|
generateAudio: {
|
|
84
87
|
type: "boolean",
|
|
85
|
-
description: "Whether the returned video should include generated/native audio. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. Supported by LTX and MiniMax H3; ignored by audio-less WAN.",
|
|
88
|
+
description: "Whether the returned video should include generated/native audio. Omit to include audio by default; set false only when the user explicitly asks for silent output or no audio. Supported by LTX and MiniMax H3; ignored by audio-less WAN.\n\nWan 3 supports this toggle; omit it for audio-on by default or set false only for an explicitly silent result.",
|
|
86
89
|
},
|
|
87
90
|
duration: {
|
|
88
91
|
type: "number",
|
|
89
|
-
description: 'Video duration in seconds. Default: 5. Per-model
|
|
92
|
+
description: 'Video duration in seconds. Default: 5. Use when the user explicitly requests a specific length (e.g., "make a 10 second video"). Per-model maximum: ltx25 and ltx23 = 20s, wan22 = 10s (clips longer than this are invalid), wan3.0-video = 30s with a 2s minimum, minimax-h3 = 15.08s with a 5.17s minimum because H3 renders 124-362 frames on a 17-frame grid at a fixed 24 fps. For totals beyond the per-model cap, batch multiple clips via sourceImageIndices instead of requesting a single oversized clip.',
|
|
93
|
+
},
|
|
94
|
+
smartDuration: {
|
|
95
|
+
type: "boolean",
|
|
96
|
+
description: "Wan 3 only. Let the model choose 2-30 seconds. Do not also set duration. The 30-second maximum is reserved and the final charge settles down to reported duration.",
|
|
97
|
+
},
|
|
98
|
+
ratio: {
|
|
99
|
+
type: "string",
|
|
100
|
+
enum: ["adaptive", "16:9", "4:3", "1:1", "3:4", "9:16"],
|
|
101
|
+
description: 'Wan 3 only. Use "adaptive" to preserve the source frame shape.',
|
|
102
|
+
},
|
|
103
|
+
watermark: {
|
|
104
|
+
type: "boolean",
|
|
105
|
+
description: "Wan 3 only. Add Alibaba's visible watermark. Defaults to false.",
|
|
90
106
|
},
|
|
91
107
|
targetResolution: {
|
|
92
108
|
type: "number",
|
|
93
|
-
description: 'Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as "480p", "720p", or "1080p" without exact pixels or an output orientation. This preserves the source image aspect ratio. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says "720p portrait" or "720p landscape", use exact-pixel aspectRatio instead.',
|
|
109
|
+
description: 'Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as "480p", "720p", or "1080p" without exact pixels or an output orientation. This preserves the source image aspect ratio. Wan 3 supports 480p, 720p, and 1080p; HappyHorse supports only 720p and 1080p. Never set 4K for either. MiniMax H3 renders inside a 1344x768 pixel budget on a 32px grid, so use 768 for H3 and never 1080p or 4K. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says "720p portrait" or "720p landscape", use exact-pixel aspectRatio instead.',
|
|
94
110
|
},
|
|
95
111
|
sourceImageIndex: {
|
|
96
112
|
type: "number",
|
|
@@ -159,4 +175,6 @@ BATCH VARIATIONS: When numberOfVariations > 1, use Dynamic Prompt syntax to vary
|
|
|
159
175
|
},
|
|
160
176
|
},
|
|
161
177
|
};
|
|
178
|
+
exports.definition.function.description +=
|
|
179
|
+
' Wan 3 first-frame and first+last-frame generation is supported with videoModel="wan3.0-video".';
|
|
162
180
|
//# sourceMappingURL=definition.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"definition.js","sourceRoot":"","sources":["../../../../src/tools/definitions/animate-photo/definition.ts"],"names":[],"mappings":";;;AAMA,kFAGiD;AACjD,sDAAmE;AACnE,kEAKsC;AAStC,MAAM,iBAAiB,GAAG;IACxB,gBAAgB;IAChB,sBAAsB;IACtB,kBAAkB;IAClB,wBAAwB;CAChB,CAAC;
|
|
1
|
+
{"version":3,"file":"definition.js","sourceRoot":"","sources":["../../../../src/tools/definitions/animate-photo/definition.ts"],"names":[],"mappings":";;;AAMA,kFAGiD;AACjD,sDAAmE;AACnE,kEAKsC;AAStC,MAAM,iBAAiB,GAAG;IACxB,gBAAgB;IAChB,sBAAsB;IACtB,kBAAkB;IAClB,wBAAwB;CAChB,CAAC;AAEX,MAAM,yBAAyB,GAC7B,2tBAA2tB,CAAC;AAEjtB,QAAA,UAAU,GAAmB;IACxC,IAAI,EAAE,UAAU;IAChB,QAAQ,EAAE;QACR,IAAI,EAAE,eAAe;QACrB,WAAW,EACT,2+OAA2+O;QAC7+O,UAAU,EAAE;YACV,IAAI,EAAE,QAAQ;YACd,UAAU,EAAE;gBACV,MAAM,EAAE;oBACN,IAAI,EAAE,QAAQ;oBACd,WAAW,EAAE;;EAErB,oDAA6B;;;;;;;;;;;;;;;;;;;;;;;;;;;;qbA4BsZ;iBAC5a;gBACD,YAAY,EAAE;oBACZ,IAAI,EAAE,SAAS;oBACf,WAAW,EACT,4GAA4G;iBAC/G;gBACD,oBAAoB,EAAE;oBACpB,IAAI,EAAE,SAAS;oBACf,WAAW,EAAE,uEAAgD;iBAC9D;gBACD,UAAU,EAAE;oBACV,IAAI,EAAE,QAAQ;oBACd,IAAI,EAAE;wBACJ,OAAO;wBACP,OAAO;wBACP,OAAO;wBACP,oBAAoB;wBACpB,oBAAoB;wBACpB,gBAAgB;wBAChB,sBAAsB;wBACtB,kBAAkB;wBAClB,wBAAwB;wBACxB,cAAc;qBACf;oBACD,WAAW,EACT,+RAA+R;wBAC/R,oqBAAoqB;wBACpqB,yBAAyB;iBAC5B;gBACD,cAAc,EAAE;oBACd,IAAI,EAAE,QAAQ;oBACd,WAAW,EACT,0QAA0Q;iBAC7Q;gBACD,aAAa,EAAE;oBACb,IAAI,EAAE,SAAS;oBACf,WAAW,EACT,+VAA+V;iBAClW;gBACD,QAAQ,EAAE;oBACR,IAAI,EAAE,QAAQ;oBACd,WAAW,EACT,qfAAqf;iBACxf;gBACD,aAAa,EAAE;oBACb,IAAI,EAAE,SAAS;oBACf,WAAW,EACT,oKAAoK;iBACvK;gBACD,KAAK,EAAE;oBACL,IAAI,EAAE,QAAQ;oBACd,IAAI,EAAE,CAAC,UAAU,EAAE,MAAM,EAAE,KAAK,EAAE,KAAK,EAAE,KAAK,EAAE,MAAM,CAAC;oBACvD,WAAW,EAAE,gEAAgE;iBAC9E;gBACD,SAAS,EAAE;oBACT,IAAI,EAAE,SAAS;oBACf,WAAW,EAAE,iEAAiE;iBAC/E;gBACD,gBAAgB,EAAE;oBAChB,IAAI,EAAE,QAAQ;oBACd,WAAW,EACT,ymBAAymB;iBAC5mB;gBACD,gBAAgB,EAAE;oBAChB,IAAI,EAAE,QAAQ;oBACd,WAAW,EACT,8bAA8b;iBACjc;gBACD,kBAAkB,EAAE;oBAClB,IAAI,EAAE,OAAO;oBACb,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE;oBACzB,QAAQ,EAAE,CAAC;oBACX,QAAQ,EAAE,EAAE;oBACZ,WAAW,EACT,qqGAAqqG;iBACxqG;gBACD,OAAO,EAAE;oBACP,IAAI,EAAE,OAAO;oBACb,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE;oBACzB,QAAQ,EAAE,CAAC;oBACX,QAAQ,EAAE,EAAE;oBACZ,WAAW,EACT,2hEAA2hE;iBAC9hE;gBACD,kBAAkB,EAAE;oBAClB,IAAI,EAAE,QAAQ;oBACd,WAAW,EACT,2XAA2X;oBAC7X,OAAO,EAAE,CAAC;oBACV,OAAO,EAAE,EAAE;iBACZ;gBACD,WAAW,EAAE;oBACX,IAAI,EAAE,QAAQ;oBACd,WAAW,EAAE,mCAAwB;iBACtC;gBACD,SAAS,EAAE;oBACT,IAAI,EAAE,QAAQ;oBACd,IAAI,EAAE,CAAC,OAAO,EAAE,KAAK,EAAE,MAAM,CAAC;oBAC9B,WAAW,EACT,yaAAya;iBAC5a;gBACD,aAAa,EAAE;oBACb,IAAI,EAAE,QAAQ;oBACd,WAAW,EACT,ykBAAykB;iBAC5kB;gBACD,eAAe,EAAE;oBACf,IAAI,EAAE,OAAO;oBACb,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE;oBACzB,QAAQ,EAAE,CAAC;oBACX,QAAQ,EAAE,EAAE;oBACZ,WAAW,EACT,whCAAwhC;iBAC3hC;gBACD,gBAAgB,EAAE;oBAChB,IAAI,EAAE,QAAQ;oBACd,WAAW,EACT,6lBAA6lB;iBAChmB;gBACD,KAAK,EAAE;oBACL,IAAI,EAAE,OAAO;oBACb,QAAQ,EAAE,CAAC;oBACX,QAAQ,EAAE,CAAC;oBACX,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,SAAS,EAAE,CAAC,EAAE;oBACvC,WAAW,EACT,sJAAsJ,wCAAsB,OAAO,IAAA,qCAAmB,EAAC,iBAAiB,CAAC,OAAO,iDAA+B,EAAE;iBACpQ;gBACD,aAAa,EAAE;oBACb,IAAI,EAAE,OAAO;oBACb,QAAQ,EAAE,CAAC;oBACX,QAAQ,EAAE,CAAC;oBACX,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE;oBACzB,WAAW,EAAE,kDAAgC;iBAC9C;aACF;YACD,QAAQ,EAAE,CAAC,QAAQ,CAAC;SACrB;KACF;CACF,CAAC;AAEF,kBAAU,CAAC,QAAQ,CAAC,WAAW;IAC7B,iGAAiG,CAAC"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"definition.d.ts","sourceRoot":"","sources":["../../../../src/tools/definitions/generate-video/definition.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AA+BlD,eAAO,MAAM,UAAU,EAAE,
|
|
1
|
+
{"version":3,"file":"definition.d.ts","sourceRoot":"","sources":["../../../../src/tools/definitions/generate-video/definition.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,aAAa,CAAC;AA+BlD,eAAO,MAAM,UAAU,EAAE,cA2LxB,CAAC"}
|