@kolbo/mcp 1.99.1 → 1.101.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kolbo/mcp",
3
- "version": "1.99.1",
3
+ "version": "1.101.0",
4
4
  "description": "Kolbo AI MCP Server - Generate images, videos, music, speech, and sound effects from Claude Code",
5
5
  "main": "src/index.js",
6
6
  "bin": {
package/src/apps/index.js CHANGED
@@ -612,11 +612,19 @@ function sibling(models, hit, types) {
612
612
  * unchanged, so identifiers the catalog does not publish (hidden models) still
613
613
  * reach the API and it stays the source of truth.
614
614
  */
615
+ const NATIVE_WORKFLOW_MODEL_IDS = new Set([
616
+ 'kolbo-morphious-motion', 'kolbo-morphious-swap', 'kolbo-morphious-reframe',
617
+ 'kolbo-morphious-lite-motion', 'kolbo-morphious-lite-swap',
618
+ 'higgsfield-genjutsu-motion-transfer', 'higgsfield-genjutsu-object-swap',
619
+ ]);
615
620
  async function canonicalModelId(client, input, type) {
616
621
  if (!input || typeof input !== 'string') return input;
617
622
  const key = input.toLowerCase().trim();
618
623
  const want = normId(key);
619
- if (!want || AUTO_ALIASES.has(want)) return input;
624
+ if (!want || AUTO_ALIASES.has(want)) return input;
625
+ // Exact native-workflow ids pass through: the near-miss check below would reject an
626
+ // unpublished or cached-out 'kolbo-*' id as unknown.
627
+ if (NATIVE_WORKFLOW_MODEL_IDS.has(key)) return input;
620
628
 
621
629
  let all;
622
630
  try {
@@ -33,6 +33,8 @@ const READ_ONLY = [
33
33
  'get_music_track_related', 'get_music_track_lyrics',
34
34
  'search_stock_media', 'get_stock_sources', 'get_stock_categories',
35
35
  'get_stock_collections', 'get_stock_asset', 'analyze_script_for_stock',
36
+ // Free (no credits, no DB write) — returns a scratch composition plan for the caller to edit.
37
+ 'create_music_composition_plan',
36
38
  ];
37
39
 
38
40
  const OPEN_WORLD_READ_ONLY = [
@@ -85,6 +87,9 @@ const DESTRUCTIVE_WRITE = [
85
87
  'generate_image', 'generate_image_edit', 'generate_creative_director',
86
88
  'generate_video', 'generate_video_from_image', 'generate_music',
87
89
  'generate_speech', 'generate_sound', 'cancel_generation',
90
+ // Both spend credits — reference-audio upload is billed by ElevenLabs like a generation,
91
+ // section edit creates a new billed track.
92
+ 'create_music_reference_audio', 'edit_music_section',
88
93
  'generate_elements', 'generate_first_last_frame', 'generate_lipsync',
89
94
  'generate_video_from_video', 'transcribe_audio', 'generate_3d',
90
95
  'edit_image', 'edit_video', 'trim_video', 'clone_voice',
@@ -57,6 +57,27 @@ const CINEMATIC_SCHEMA = z.object({
57
57
  // alongside the prompt (generate_video_from_image: `{ image_url, prompt }` — the
58
58
  // image is what varies, and that is the whole point). Either way the widget
59
59
  // captions each tile with the item's prompt, so that is the label we carry.
60
+ // ElevenLabs Music v2.5 composition-plan chunk — one entry is EITHER a generation chunk
61
+ // (text/duration_ms/positive_styles, optionally conditioning_ref to steer its style off a
62
+ // stored song) OR an audio-reference chunk (song_id/range, splices that slice in unchanged).
63
+ // All fields optional at the schema level; the backend validates the real constraints
64
+ // (3-120s per generation chunk, <=30s conditioning reference, <=30 chunks per plan).
65
+ const MUSIC_TIME_RANGE_SCHEMA = z.object({
66
+ start_ms: z.number().int().min(0),
67
+ end_ms: z.number().int().min(0),
68
+ }).describe('{ start_ms, end_ms } — a slice of a stored song, in milliseconds.');
69
+ const MUSIC_CHUNK_SCHEMA = z.object({
70
+ text: z.string().optional().describe('Generation chunk only: section name in [brackets], lyric lines, {inline directions}. E.g. "[Chorus]\\nWe rise tonight".'),
71
+ duration_ms: z.number().int().min(3000).max(120000).optional().describe('Generation chunk only: length in ms, 3000-120000.'),
72
+ positive_styles: z.array(z.string()).optional().describe('Generation chunk only: styles/directions to include (max 50). The first chunk\'s styles set the whole song\'s tone.'),
73
+ negative_styles: z.array(z.string()).optional().describe('Generation chunk only: styles/directions to avoid.'),
74
+ context_adherence: z.enum(['low', 'medium', 'high']).optional().describe('Generation chunk only: how closely it follows neighboring chunks. Default high.'),
75
+ conditioning_ref: z.object({ song_id: z.string(), range: MUSIC_TIME_RANGE_SCHEMA }).optional().describe('Generation chunk only: condition this chunk\'s STYLE on <=30s of a stored song (does not splice audio in — it steers a fresh generation).'),
76
+ condition_strength: z.enum(['low', 'medium', 'high', 'xhigh']).optional().describe('Generation chunk only: how strongly it follows conditioning_ref. Default medium.'),
77
+ song_id: z.string().optional().describe('Audio-reference chunk only: id of a stored song (from a prior generate_music result\'s song_id, or create_music_reference_audio) to splice in unchanged.'),
78
+ range: MUSIC_TIME_RANGE_SCHEMA.optional().describe('Audio-reference chunk only: the slice of song_id to insert unchanged.'),
79
+ });
80
+
60
81
  const MAX_BATCH_PROMPTS = 8;
61
82
  const MAX_STATUS_IDS = 20;
62
83
  async function submitBatch(rawItems, submitOne) {
@@ -212,6 +233,15 @@ const promptsField = (what) => z.array(z.string()).max(MAX_BATCH_PROMPTS).option
212
233
  `BATCH MODE — several DIFFERENT prompts (2–${MAX_BATCH_PROMPTS}) generated concurrently in ONE call and rendered together in ONE combined widget. **Hard cap: ${MAX_BATCH_PROMPTS} prompts per call — more than that is REJECTED with an error (never silently truncated), so split a longer list across several calls of at most ${MAX_BATCH_PROMPTS}.** Whenever the user wants multiple distinct ${what} with their own prompts, ALWAYS pass them all here instead of making several separate calls — separate calls clutter the chat with stacked widgets. All prompts share the same model/settings. When set, \`prompt\` is ignored. For N variations of a SINGLE prompt use num_images (image tools); for an AI-planned coherent scene set use generate_creative_director.`
213
234
  );
214
235
 
236
+ // Native workflows: exactly one source video, prompt optional (the server analyzes the media).
237
+ const MORPHIOUS_MODELS = new Set(['kolbo-morphious-motion', 'kolbo-morphious-swap', 'kolbo-morphious-reframe', 'kolbo-morphious-lite-motion', 'kolbo-morphious-lite-swap']);
238
+ const PROMPTLESS_NATIVE = new Set([...MORPHIOUS_MODELS, 'higgsfield-genjutsu-motion-transfer', 'higgsfield-genjutsu-object-swap']);
239
+ const MORPHIOUS_GUIDE = ' MORPHIOUS / GENJUTSU (exactly ONE source video, omit duration - it comes from the source): higgsfield-genjutsu-motion-transfer and higgsfield-genjutsu-object-swap accept an omitted prompt, and so do the kolbo-morphious-* models (Morphious only needs the source video and references; it analyzes them itself). '
240
+ + 'kolbo-morphious-motion 1-30 images, 4-30s; kolbo-morphious-swap 0-30 images, 4-30s; both 480p/720p/1080p, output keeps the source shape. '
241
+ + 'Lite (kolbo-morphious-lite-motion 1-10 images / kolbo-morphious-lite-swap 0-10 images) caps at 15s. '
242
+ + 'kolbo-morphious-reframe: NO images/DNA, 1-300s, aspect_ratio REQUIRED (16:9, 9:16, 1:1, 4:3, 3:4, 21:9), 540p or 720p. '
243
+ + 'higgsfield-genjutsu-motion-transfer / -object-swap: 1-8 images or Visual DNA, 4-30s, 480p/720p/1080p.';
244
+
215
245
  function registerGenerateTools(server, client, options = {}) {
216
246
  // Every JSON POST from these tools rehosts local file paths into the media
217
247
  // library first (see local-rehost.js) — the API only understands URLs.
@@ -797,7 +827,7 @@ function registerGenerateTools(server, client, options = {}) {
797
827
  'generate_music',
798
828
  'Generate music from a text description using Kolbo AI. Supports instrumental mode, custom lyrics, style direction, vocal gender, negative tags, song length, and Suno fine-controls (style weight, weirdness, audio weight, persona/singing voice). Default model is Suno. Some controls are Suno-only; the engine ignores controls that do not apply to the chosen model. Returns the final audio URL when complete.',
799
829
  {
800
- prompt: z.string().describe('Text description of the music to generate (e.g., "upbeat electronic dance track with synthesizers")'),
830
+ prompt: z.string().optional().describe('Text description of the music to generate (e.g., "upbeat electronic dance track with synthesizers"). Required unless composition_plan is passed (ElevenLabs Music v2.5 structured mode, which cannot be combined with prompt).'),
801
831
  model: z.string().optional().describe('Model identifier. Use list_models type="music_gen" to see options. Omit for Suno (default).'),
802
832
  style: z.string().optional().describe('Music style / genre (e.g., "pop", "rock", "lo-fi", "electronic", "jazz")'),
803
833
  title: z.string().optional().describe('Song title. If omitted, one is generated.'),
@@ -816,16 +846,22 @@ function registerGenerateTools(server, client, options = {}) {
816
846
  use_composition_plan: z.boolean().optional().describe('Suno: enable structured composition planning (verse/chorus structure).'),
817
847
  singing_dna_id: z.string().optional().describe('Visual DNA character id whose singing voice to use (must be owned by the caller).'),
818
848
  singing_voice_id: z.string().optional().describe('Custom cloned singing-voice id (must be owned by the caller).'),
849
+ // ── ElevenLabs Music v2.5 structured mode (mutually exclusive with `prompt`) ──
850
+ composition_plan: z.array(MUSIC_CHUNK_SCHEMA).optional().describe('ElevenLabs Music v2.5 ONLY, model must resolve to the ElevenLabs music model: an ordered list of chunks instead of a prompt. A GENERATION chunk has text/duration_ms/positive_styles (optionally conditioning_ref+condition_strength to steer it off a stored song). An AUDIO-REFERENCE chunk has song_id/range and splices that stored slice in unchanged (used to keep part of an existing track — see edit_music_section for the common case of regenerating one section). Get a starter plan from create_music_composition_plan, or build one from a prior generate_music result\'s `song_id`. Cannot be combined with `prompt`.'),
851
+ seed: z.number().optional().describe('ElevenLabs Music v2.5: random seed for reproducibility. Only valid together with composition_plan, never with prompt.'),
819
852
  project_id: projectIdField,
820
853
  session_id: sessionIdField
821
854
  },
822
- async ({ prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags, duration_seconds, enhance_prompt = false, preset_id, style_weight, weirdness, audio_weight, persona_id, use_composition_plan, singing_dna_id, singing_voice_id, project_id, session_id }) => {
855
+ async ({ prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags, duration_seconds, enhance_prompt = false, preset_id, style_weight, weirdness, audio_weight, persona_id, use_composition_plan, singing_dna_id, singing_voice_id, composition_plan, seed, project_id, session_id }) => {
856
+ if (!prompt && !composition_plan) throw new Error('prompt is required (or pass composition_plan for ElevenLabs Music v2.5)');
823
857
  model = await canonicalModelId(client, model, 'music_gen'); // lenient id resolution ("z-image" → "z-image/turbo")
824
858
  const gen = await client.post('/v1/generate/music', {
825
859
  prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags,
826
860
  duration_seconds, enhance_prompt, preset_id,
827
861
  style_weight, weirdness, audio_weight, persona_id, use_composition_plan,
828
- singing_dna_id, singing_voice_id, project_id, session_id
862
+ singing_dna_id, singing_voice_id,
863
+ ...(composition_plan ? { composition_plan: { chunks: composition_plan }, seed } : {}),
864
+ project_id, session_id
829
865
  });
830
866
 
831
867
  if (returnsImmediately()) return submittedResult({
@@ -855,11 +891,99 @@ function registerGenerateTools(server, client, options = {}) {
855
891
  playback_urls: result.result.playback_urls,
856
892
  title: result.result.title,
857
893
  duration: result.result.duration,
858
- lyrics: result.result.lyrics
894
+ lyrics: result.result.lyrics,
895
+ // ElevenLabs Music v2.5 only — reuse in a later composition_plan / edit_music_section.
896
+ ...(result.result.song_id ? { song_id: result.result.song_id } : {}),
859
897
  }, null, 2));
860
898
  }
861
899
  );
862
900
 
901
+ // ─── create_music_composition_plan ─────────────────────────
902
+ // FREE (no credits, no generation created) — ElevenLabs Music v2.5's POST /v1/music/plan.
903
+ server.tool(
904
+ 'create_music_composition_plan',
905
+ 'Generate a starter ElevenLabs Music v2.5 structured composition plan (an ordered list of chunks with section text, styles, and durations) from a plain-text prompt. FREE — creates no generation and charges no credits. Edit the returned chunks and pass them as `composition_plan` to generate_music to actually create the track.',
906
+ {
907
+ prompt: z.string().describe('Text description of the song to plan (e.g. "an upbeat pop song with verse and chorus about summer").'),
908
+ duration_seconds: z.number().optional().describe('Target total length in seconds, 3-600. Omit to let the model choose.'),
909
+ project_id: projectIdField,
910
+ session_id: sessionIdField,
911
+ },
912
+ async ({ prompt, duration_seconds, project_id, session_id }) => {
913
+ const res = await client.post('/v1/generate/music/composition-plan', { prompt, duration_seconds, project_id, session_id });
914
+ return { content: [{ type: 'text', text: JSON.stringify(res, null, 2) }] };
915
+ }
916
+ );
917
+
918
+ // ─── create_music_reference_audio ───────────────────────────
919
+ // BILLED at the same rate as a song generation of the clip's length (ElevenLabs charges
920
+ // the upload itself, even half-price if it flags the clip for copyright) — this is NOT a
921
+ // free utility call, unlike create_music_composition_plan above.
922
+ server.tool(
923
+ 'create_music_reference_audio',
924
+ 'Upload an existing audio clip you already have in Kolbo so it can be used as an audio reference in a later generate_music composition_plan (as a song_id in a conditioning_ref, to steer style, or in an audio-reference chunk, to splice a slice in unchanged) or as the source for edit_music_section. BILLED the same as generating a track of the clip\'s length — ElevenLabs charges for this upload itself. LOCAL FILE? Call upload_media first and pass the returned media id here.',
925
+ {
926
+ media_id: z.string().describe('A media-library item id for an audio file you own — from upload_media, or from list_media (mediaType audio).'),
927
+ project_id: projectIdField,
928
+ session_id: sessionIdField,
929
+ },
930
+ async ({ media_id, project_id, session_id }) => {
931
+ const res = await client.post('/v1/generate/music/reference-audio', { media_id, project_id, session_id });
932
+ return { content: [{ type: 'text', text: JSON.stringify(res, null, 2) }] };
933
+ }
934
+ );
935
+
936
+ // ─── edit_music_section ─────────────────────────────────────
937
+ // ElevenLabs Music v2.5 inpainting — regenerate one time range of an existing track,
938
+ // keeping the rest unchanged. BILLED like an ordinary generation (creates a new track).
939
+ server.tool(
940
+ 'edit_music_section',
941
+ 'Regenerate one section (time range) of an ElevenLabs Music v2.5 track you already generated, keeping the rest of the song unchanged — e.g. "redo the outro with different lyrics". Only works on tracks made with the ElevenLabs Music model (check the earlier generate_music result for a `song_id`; a track without one cannot be edited this way). Creates a NEW generation in the same session; the source track is untouched. Returns a generation_id to poll like any other generate_music call.',
942
+ {
943
+ source_generation_id: z.string().describe('The generation_id of the ElevenLabs Music v2.5 track to edit (must have returned a song_id).'),
944
+ start_ms: z.number().int().min(0).describe('Start of the region to regenerate, in milliseconds.'),
945
+ end_ms: z.number().int().min(0).describe('End of the region to regenerate, in milliseconds. (end_ms - start_ms) must be 3000-120000ms.'),
946
+ text: z.string().optional().describe('New section text — [Section Name], lyric lines, {inline directions}. Omit to keep it instrumental/unlabeled.'),
947
+ positive_styles: z.array(z.string()).optional().describe('Styles/directions for the regenerated section.'),
948
+ negative_styles: z.array(z.string()).optional().describe('Styles/directions to avoid in the regenerated section.'),
949
+ context_adherence: z.enum(['low', 'medium', 'high']).optional().describe('How closely the new section follows the kept audio around it. Default high.'),
950
+ project_id: projectIdField,
951
+ session_id: sessionIdField,
952
+ },
953
+ async ({ source_generation_id, start_ms, end_ms, text, positive_styles, negative_styles, context_adherence, project_id, session_id }) => {
954
+ const gen = await client.post('/v1/generate/music/edit-section', {
955
+ source_generation_id, start_ms, end_ms, text,
956
+ positive_styles, negative_styles, context_adherence,
957
+ project_id, session_id,
958
+ });
959
+
960
+ if (returnsImmediately()) return submittedResult({
961
+ tool: 'edit_music_section', kind: 'audio', gen, client, model: 'ElevenLabs Music', prompt: text || '(section edit)',
962
+ });
963
+
964
+ const poll = await pollOrTimedOut(client, gen.generation_id, { interval: (gen.poll_interval_hint || 8) * 1000, timeout: 150000 });
965
+ if (poll.timedOut) return poll.timedOut;
966
+ const result = poll.result;
967
+
968
+ return uiCompleted({
969
+ tool: 'edit_music_section', kind: 'audio', gen, client, model: 'ElevenLabs Music', prompt: text || '(section edit)',
970
+ urls: result.result.urls,
971
+ playback_urls: result.result.playback_urls,
972
+ title: result.result.title,
973
+ duration: result.result.duration,
974
+ credits_used: creditFields(result).credits_used,
975
+ }, JSON.stringify({
976
+ ...creditFields(result),
977
+ session_id: gen.session_id,
978
+ urls: result.result.urls,
979
+ playback_urls: result.result.playback_urls,
980
+ duration: result.result.duration,
981
+ ...(result.result.song_id ? { song_id: result.result.song_id } : {}),
982
+ }, null, 2));
983
+ }
984
+ );
985
+
986
+
863
987
  // ─── music import / extend / cover ─────────────────────────
864
988
  // Three tools over the SDK's /v1/generate/music/{import,extend,cover}. The web app has
865
989
  // had upload-extend and upload-cover for a long time; agents could not reach either
@@ -1434,9 +1558,9 @@ function registerGenerateTools(server, client, options = {}) {
1434
1558
  // ─── generate_elements ─────────────────────────────────────
1435
1559
  server.tool(
1436
1560
  'generate_elements',
1437
- 'Generate a video from reference elements (images, videos, and/or audio) + a text prompt. SPECIAL MODEL EXCEPTION: seedance-2-5-multilingual accepts an ordinary prompt in any language with the ORIGINAL target-language dialogue in double quotes, including Hebrew script. Use this tool even without references for that model, duration 8 or 12, resolution 480p/720p/1080p (no Draft), up to 3 still references or Visual DNAs, no video/audio references. It performs translation, distinct designed voices and segmented lip sync internally; never transliterate or pre-translate its quoted dialogue. The following regular Seedance instructions apply only to other model IDs. Use when the user wants to animate specific uploaded/referenced assets — e.g. "animate this product", "put these 3 characters into a scene". PRIMARY ROUTE FOR A DNA-ANCHORED MULTI-SHOT FILM: one call can carry the whole sequence — seedance-2-5 takes 4-30s, up to 30 shots and 20 Visual DNAs in a SINGLE generation (seedance-2: 4-15s, 9 DNAs) — instead of a stack of separate clips. DIALOGUE IS PERFORMED NATIVELY: quoted dialogue in the prompt comes back as synced voices with lip movement, room tone and the SFX named in the AUDIO block — never route scene dialogue to generate_speech or generate_lipsync. Write dialogue in ENGLISH or Latin transliteration of Hebrew ("shalom"), never Hebrew script — Seedance does not speak Hebrew; prefer Gemini Omni Flash 1.1 or Gemini Omni 1 for native Hebrew. COST: resolution is a multiplier. When list_models publishes `video_input_credit` and this call carries videos, charge that rate against nominal input seconds + nominal output seconds; MP4 padding within 0.15s of an integer snaps to that integer and larger fractions round up. Otherwise use the normal output-second rate. PROMPT CONTRACT (Seedance / Elements): Locked Intro only — Total line, then [GLOBAL LOOK] / [CAST] / [LOCATION] / SHOT N. Do NOT write SCENE CONTEXT / OPTICS / ACTION department packs. Every Visual DNA in visual_dna_ids MUST also appear in the prompt as @ExactDNAName (e.g. "@Zohar walks…") — never "Zohar\'s" or "the man on the left" as a substitute. IMPORTANT: different models accept different numbers and durations of inputs — call list_models type="elements" and read elements_max_images / elements_max_videos / elements_max_audio plus min_video_duration / max_video_duration before generating. For text-only → video use generate_video instead. For animating a single still image use generate_video_from_image. Returns the final video URL when complete.',
1561
+ 'Generate a video from reference elements (images, videos, and/or audio) + a text prompt. SPECIAL MODEL EXCEPTION: seedance-2-5-multilingual accepts an ordinary prompt in any language with the ORIGINAL target-language dialogue in double quotes, including Hebrew script. Use this tool even without references for that model, duration 8 or 12, resolution 480p/720p/1080p (no Draft), up to 3 still references or Visual DNAs, no video/audio references. It performs translation, distinct designed voices and segmented lip sync internally; never transliterate or pre-translate its quoted dialogue. The following regular Seedance instructions apply only to other model IDs. Use when the user wants to animate specific uploaded/referenced assets — e.g. "animate this product", "put these 3 characters into a scene". PRIMARY ROUTE FOR A DNA-ANCHORED MULTI-SHOT FILM: one call can carry the whole sequence — seedance-2-5 takes 4-30s, up to 30 shots and 20 Visual DNAs in a SINGLE generation (seedance-2: 4-15s, 9 DNAs) — instead of a stack of separate clips. DIALOGUE IS PERFORMED NATIVELY: quoted dialogue in the prompt comes back as synced voices with lip movement, room tone and the SFX named in the AUDIO block — never route scene dialogue to generate_speech or generate_lipsync. Write dialogue in ENGLISH or Latin transliteration of Hebrew ("shalom"), never Hebrew script — Seedance does not speak Hebrew; prefer Gemini Omni Flash 1.1 or Gemini Omni 1 for native Hebrew. COST: resolution is a multiplier. When list_models publishes `video_input_credit` and this call carries videos, charge that rate against nominal input seconds + nominal output seconds; MP4 padding within 0.15s of an integer snaps to that integer and larger fractions round up. Otherwise use the normal output-second rate. PROMPT CONTRACT (Seedance / Elements): Locked Intro only — Total line, then [GLOBAL LOOK] / [CAST] / [LOCATION] / SHOT N. Do NOT write SCENE CONTEXT / OPTICS / ACTION department packs. Every Visual DNA in visual_dna_ids MUST also appear in the prompt as @ExactDNAName (e.g. "@Zohar walks…") — never "Zohar\'s" or "the man on the left" as a substitute. IMPORTANT: different models accept different numbers and durations of inputs — call list_models type="elements" and read elements_max_images / elements_max_videos / elements_max_audio plus min_video_duration / max_video_duration before generating. For text-only → video use generate_video instead. For animating a single still image use generate_video_from_image. Returns the final video URL when complete.' + MORPHIOUS_GUIDE,
1438
1562
  {
1439
- prompt: z.string().describe('For seedance-2-5-multilingual: ordinary scene description with original target-language dialogue in double quotes; no manual translation or transliteration. For other models, Locked Intro prompt (Seedance/Elements): Total line, [GLOBAL LOOK], [CAST] with @ExactDNAName for every visual_dna_ids entry, [LOCATION], then SHOT N. Not SCENE CONTEXT/OPTICS/ACTION packs. Never substitute "the left man" or "Zohar\'s" for @Name. EVERY attached reference must also be tagged by its 1-based array position — `@Image 1`/`@Image 2` (reference_images), `@Video 1` (reference_videos), `@Audio 1` (reference_audio_urls) — and its job stated ("@Image 1 defines the character\'s face and wardrobe", "@Video 1 defines the camera move"). An untagged attachment is ignored by the engine even though it was uploaded and billed.'),
1563
+ prompt: z.string().optional().describe('Optional ONLY for kolbo-morphious-* and higgsfield-genjutsu-* models (the server analyzes the media). For seedance-2-5-multilingual: ordinary scene description with original target-language dialogue in double quotes; no manual translation or transliteration. For other models, Locked Intro prompt (Seedance/Elements): Total line, [GLOBAL LOOK], [CAST] with @ExactDNAName for every visual_dna_ids entry, [LOCATION], then SHOT N. Not SCENE CONTEXT/OPTICS/ACTION packs. Never substitute "the left man" or "Zohar\'s" for @Name. EVERY attached reference must also be tagged by its 1-based array position — `@Image 1`/`@Image 2` (reference_images), `@Video 1` (reference_videos), `@Audio 1` (reference_audio_urls) — and its job stated ("@Image 1 defines the character\'s face and wardrobe", "@Video 1 defines the camera move"). An untagged attachment is ignored by the engine even though it was uploaded and billed.'),
1440
1564
  model: z.string().optional().describe('Model identifier. If the user already named a family (Grok / Kling / Veo / Seedance / …), pass THAT family — never default to Seedance because Elements often uses it. Use list_models type="elements" for exact ids and elements_max_* caps. Do NOT omit (omitting = Smart Select).'),
1441
1565
  reference_images: z.array(z.string()).optional().describe('Array of image references (product shots, character references, etc.). Tag each one in the prompt text by its 1-based position here — item 1 is `@Image 1`, item 2 is `@Image 2` — or the engine ignores it. Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_images` URLs from list_models for the chosen model — exceeding it is a deterministic 400.**'),
1442
1566
  reference_videos: z.array(z.string()).optional().describe('Array of reference videos for models that accept video inputs. Tag each one in the prompt text by its 1-based position here — `@Video 1`, `@Video 2` — stating what it defines (motion, camera move, pacing), or the engine ignores it. Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_videos` URLs and keep every clip within `min_video_duration`-`max_video_duration` from list_models.** If `video_input_credit` is present, every attached video contributes its nominal duration to combined-second billing: encoder padding within 0.15s of an integer snaps to it; larger fractions round up.'),
@@ -1462,14 +1586,14 @@ function registerGenerateTools(server, client, options = {}) {
1462
1586
  project_id: projectIdField,
1463
1587
  session_id: sessionIdField
1464
1588
  },
1465
- async ({ prompt, model, reference_images, reference_videos, reference_audio_urls, audio_url, files, duration, aspect_ratio, motion, preset_id, enhance_prompt = false, visual_dna_ids, resolution, draft, sound_enabled, keyframes, multi_shots, multi_shot_count, session_name, project_id, session_id }) => {
1589
+ async ({ prompt = '', model, reference_images, reference_videos, reference_audio_urls, audio_url, files, duration, aspect_ratio, motion, preset_id, enhance_prompt = false, visual_dna_ids, resolution, draft, sound_enabled, keyframes, multi_shots, multi_shot_count, session_name, project_id, session_id }) => {
1466
1590
  validateVideoPrompt({ prompt, duration, aspect_ratio, multi_shots, multi_shot_count });
1467
1591
  if (draft === true) resolution = '480p-draft';
1468
1592
  else if (draft === false && resolution?.endsWith('-draft')) resolution = resolution.slice(0, -6);
1469
1593
  model = await canonicalModelId(client, model, 'elements'); // lenient id resolution ("z-image" → "z-image/turbo")
1470
1594
  aspect_ratio = await resolveCatalogAspectRatio(client, model, aspect_ratio, 'elements');
1471
1595
  validateVideoPrompt({ prompt, duration, aspect_ratio, multi_shots, multi_shot_count });
1472
- if (!prompt) throw new Error('prompt is required');
1596
+ if (!prompt && !PROMPTLESS_NATIVE.has(model)) throw new Error('prompt is required');
1473
1597
 
1474
1598
  // Elements is the one tool that takes all three modalities, and either a
1475
1599
  // URL or a local path for any of them. Bucket every input by the kind the
@@ -1855,10 +1979,10 @@ function registerGenerateTools(server, client, options = {}) {
1855
1979
  // ─── generate_video_from_video ─────────────────────────────
1856
1980
  server.tool(
1857
1981
  'generate_video_from_video',
1858
- 'Restyle / transform an existing video (video-to-video). Use for style transfer, scene restyling, subject swap, motion transfer, character replacement, or burning in styled subtitles (VEED Subtitles). Source video can be a URL or absolute local path. `prompt` is OPTIONAL: most models need it, but prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control) ignore it. For VEED Subtitles, pass a `preset` style and optional `source_language` / `translation_language` instead of a prompt. IMPORTANT: different models support different extra inputs — call list_models type="video_to_video" and read max_images / max_videos / max_elements on the chosen model before generating. Pass reference_images for models with max_images > 0 (e.g. Kling O1/O3, Aleph, WAN VACE), reference_videos for models with max_videos > 1 (e.g. WAN 2.6 reference-to-video accepts up to 3), and elements for models with max_elements > 0. REPAIR / RETIME (not restyling, no prompt needed — these keep the footage and fix or retime it): model "topaz/deblur/video" removes lens, motion and compression blur; "topaz/colorize/video" colorizes black-and-white footage; "topaz/interpolate/video" is SLOW MOTION and frame-rate conversion (see slowdown_factor / target_fps); "topaz/sdr-to-hdr/video" masters SDR footage to HDR (see output_format). Reach for these when the user says blurry, shaky-detail, black-and-white, slow motion, smoother frame rate, or HDR — a restyle model would repaint the video instead of repairing it. For animating a still image use generate_video_from_image instead. For text-only → video use generate_video.',
1982
+ 'Restyle / transform an existing video (video-to-video). Use for style transfer, scene restyling, subject swap, motion transfer, character replacement, or burning in styled subtitles (VEED Subtitles). Source video can be a URL or absolute local path. `prompt` is OPTIONAL: most models need it, but prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control) ignore it. For VEED Subtitles, pass a `preset` style and optional `source_language` / `translation_language` instead of a prompt. IMPORTANT: different models support different extra inputs — call list_models type="video_to_video" and read max_images / max_videos / max_elements on the chosen model before generating. Pass reference_images for models with max_images > 0 (e.g. Kling O1/O3, Aleph, WAN VACE), reference_videos for models with max_videos > 1 (e.g. WAN 2.6 reference-to-video accepts up to 3), and elements for models with max_elements > 0. REPAIR / RETIME (not restyling, no prompt needed — these keep the footage and fix or retime it): model "topaz/deblur/video" removes lens, motion and compression blur; "topaz/colorize/video" colorizes black-and-white footage; "topaz/interpolate/video" is SLOW MOTION and frame-rate conversion (see slowdown_factor / target_fps); "topaz/sdr-to-hdr/video" masters SDR footage to HDR (see output_format). Reach for these when the user says blurry, shaky-detail, black-and-white, slow motion, smoother frame rate, or HDR — a restyle model would repaint the video instead of repairing it. For animating a still image use generate_video_from_image instead. For text-only → video use generate_video.' + MORPHIOUS_GUIDE,
1859
1983
  {
1860
1984
  source_video: z.string().describe('URL or absolute local path to the primary source video to restyle. **Source duration must fall within `min_video_duration`-`max_video_duration` from list_models for the chosen model** — videos outside that range are rejected (or silently truncated by some upstream providers). For models that use reference_videos as their primary input (e.g. WAN 2.6 reference-to-video), pass the first reference video here and also include it in reference_videos.'),
1861
- prompt: z.string().optional().describe('Text description of the desired restyle / transformation. Required by most video-to-video models; omit for prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control).'),
1985
+ prompt: z.string().optional().describe('Text description of the desired restyle / transformation. Required by most video-to-video models; omit for prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control, kolbo-morphious-*, higgsfield-genjutsu-motion-transfer, higgsfield-genjutsu-object-swap).'),
1862
1986
  model: z.string().optional().describe('Model identifier. Use list_models type="video_to_video" to see options and check max_images / max_videos / max_elements / max_video_duration per model. Pick a SPECIFIC model — do NOT omit (omitting = Smart Select auto-pick, which we avoid); call list_models for this type and choose the model that best fits the user\'s intent.'),
1863
1987
  aspect_ratio: z.string().optional().describe(aspectRatioDescribe() + ' Default: matches source.'),
1864
1988
  duration: z.number().optional().describe('Output duration in seconds. Must be in `supported_durations` from list_models, OR within `min_output_duration`-`max_output_duration`. Default: matches source'),
@@ -1,6 +1,6 @@
1
1
  // Validate only explicit, machine-readable declarations. Never infer creative
2
2
  // intent from adjectives, require a template, or rewrite a user's prompt.
3
- function validateVideoPrompt({ prompt, duration, aspect_ratio, multi_shots, multi_shot_count }) {
3
+ function validateVideoPrompt({ prompt = '', duration, aspect_ratio, multi_shots, multi_shot_count }) {
4
4
  const totals = [...prompt.matchAll(/^\s*Total:\s*(\d+(?:\.\d+)?)s\s*\/\s*(\d+)\s*shots?\s*\/\s*(\d+:\d+)\s*$/gim)]
5
5
  .map((m) => ({ duration: Number(m[1]), count: Number(m[2]), aspect: m[3] }));
6
6
  if (!totals.length) return; // Existing free-form clients remain supported.