@kolbo/mcp 1.100.0 → 1.101.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kolbo/mcp",
3
- "version": "1.100.0",
3
+ "version": "1.101.1",
4
4
  "description": "Kolbo AI MCP Server - Generate images, videos, music, speech, and sound effects from Claude Code",
5
5
  "main": "src/index.js",
6
6
  "bin": {
@@ -1,6 +1,6 @@
1
1
  # AUTO-GENERATED — do not edit
2
2
 
3
- This tree is mirrored from kolbo-code@e4764c5, the single source of truth.
3
+ This tree is mirrored from kolbo-code@f1276f9, the single source of truth.
4
4
  Canonical source: packages/opencode/skills/kolbo/
5
5
  Distribution: .github/workflows/sync-skill-to-plugin.yml
6
6
 
package/src/apps/index.js CHANGED
@@ -612,11 +612,19 @@ function sibling(models, hit, types) {
612
612
  * unchanged, so identifiers the catalog does not publish (hidden models) still
613
613
  * reach the API and it stays the source of truth.
614
614
  */
615
+ const NATIVE_WORKFLOW_MODEL_IDS = new Set([
616
+ 'kolbo-morphious-motion', 'kolbo-morphious-swap', 'kolbo-morphious-reframe',
617
+ 'kolbo-morphious-lite-motion', 'kolbo-morphious-lite-swap',
618
+ 'higgsfield-genjutsu-motion-transfer', 'higgsfield-genjutsu-object-swap',
619
+ ]);
615
620
  async function canonicalModelId(client, input, type) {
616
621
  if (!input || typeof input !== 'string') return input;
617
622
  const key = input.toLowerCase().trim();
618
623
  const want = normId(key);
619
- if (!want || AUTO_ALIASES.has(want)) return input;
624
+ if (!want || AUTO_ALIASES.has(want)) return input;
625
+ // Exact native-workflow ids pass through: the near-miss check below would reject an
626
+ // unpublished or cached-out 'kolbo-*' id as unknown.
627
+ if (NATIVE_WORKFLOW_MODEL_IDS.has(key)) return input;
620
628
 
621
629
  let all;
622
630
  try {
@@ -233,6 +233,15 @@ const promptsField = (what) => z.array(z.string()).max(MAX_BATCH_PROMPTS).option
233
233
  `BATCH MODE — several DIFFERENT prompts (2–${MAX_BATCH_PROMPTS}) generated concurrently in ONE call and rendered together in ONE combined widget. **Hard cap: ${MAX_BATCH_PROMPTS} prompts per call — more than that is REJECTED with an error (never silently truncated), so split a longer list across several calls of at most ${MAX_BATCH_PROMPTS}.** Whenever the user wants multiple distinct ${what} with their own prompts, ALWAYS pass them all here instead of making several separate calls — separate calls clutter the chat with stacked widgets. All prompts share the same model/settings. When set, \`prompt\` is ignored. For N variations of a SINGLE prompt use num_images (image tools); for an AI-planned coherent scene set use generate_creative_director.`
234
234
  );
235
235
 
236
+ // Native workflows: exactly one source video, prompt optional (the server analyzes the media).
237
+ const MORPHIOUS_MODELS = new Set(['kolbo-morphious-motion', 'kolbo-morphious-swap', 'kolbo-morphious-reframe', 'kolbo-morphious-lite-motion', 'kolbo-morphious-lite-swap']);
238
+ const PROMPTLESS_NATIVE = new Set([...MORPHIOUS_MODELS, 'higgsfield-genjutsu-motion-transfer', 'higgsfield-genjutsu-object-swap']);
239
+ const MORPHIOUS_GUIDE = ' MORPHIOUS / GENJUTSU (exactly ONE source video, omit duration - it comes from the source): higgsfield-genjutsu-motion-transfer and higgsfield-genjutsu-object-swap accept an omitted prompt, and so do the kolbo-morphious-* models (Morphious only needs the source video and references; it analyzes them itself). '
240
+ + 'kolbo-morphious-motion 1-30 images, 4-30s; kolbo-morphious-swap 0-30 images, 4-30s; both 480p/720p/1080p, output keeps the source shape. '
241
+ + 'Lite (kolbo-morphious-lite-motion 1-10 images / kolbo-morphious-lite-swap 0-10 images) caps at 15s. '
242
+ + 'kolbo-morphious-reframe: NO images/DNA, 1-300s, aspect_ratio REQUIRED (16:9, 9:16, 1:1, 4:3, 3:4, 21:9), 540p or 720p. '
243
+ + 'higgsfield-genjutsu-motion-transfer / -object-swap: 1-8 images or Visual DNA, 4-30s, 480p/720p/1080p.';
244
+
236
245
  function registerGenerateTools(server, client, options = {}) {
237
246
  // Every JSON POST from these tools rehosts local file paths into the media
238
247
  // library first (see local-rehost.js) — the API only understands URLs.
@@ -1549,9 +1558,9 @@ function registerGenerateTools(server, client, options = {}) {
1549
1558
  // ─── generate_elements ─────────────────────────────────────
1550
1559
  server.tool(
1551
1560
  'generate_elements',
1552
- 'Generate a video from reference elements (images, videos, and/or audio) + a text prompt. SPECIAL MODEL EXCEPTION: seedance-2-5-multilingual accepts an ordinary prompt in any language with the ORIGINAL target-language dialogue in double quotes, including Hebrew script. Use this tool even without references for that model, duration 8 or 12, resolution 480p/720p/1080p (no Draft), up to 3 still references or Visual DNAs, no video/audio references. It performs translation, distinct designed voices and segmented lip sync internally; never transliterate or pre-translate its quoted dialogue. The following regular Seedance instructions apply only to other model IDs. Use when the user wants to animate specific uploaded/referenced assets — e.g. "animate this product", "put these 3 characters into a scene". PRIMARY ROUTE FOR A DNA-ANCHORED MULTI-SHOT FILM: one call can carry the whole sequence — seedance-2-5 takes 4-30s, up to 30 shots and 20 Visual DNAs in a SINGLE generation (seedance-2: 4-15s, 9 DNAs) — instead of a stack of separate clips. DIALOGUE IS PERFORMED NATIVELY: quoted dialogue in the prompt comes back as synced voices with lip movement, room tone and the SFX named in the AUDIO block — never route scene dialogue to generate_speech or generate_lipsync. Write dialogue in ENGLISH or Latin transliteration of Hebrew ("shalom"), never Hebrew script — Seedance does not speak Hebrew; prefer Gemini Omni Flash 1.1 or Gemini Omni 1 for native Hebrew. COST: resolution is a multiplier. When list_models publishes `video_input_credit` and this call carries videos, charge that rate against nominal input seconds + nominal output seconds; MP4 padding within 0.15s of an integer snaps to that integer and larger fractions round up. Otherwise use the normal output-second rate. PROMPT CONTRACT (Seedance / Elements): Locked Intro only — Total line, then [GLOBAL LOOK] / [CAST] / [LOCATION] / SHOT N. Do NOT write SCENE CONTEXT / OPTICS / ACTION department packs. Every Visual DNA in visual_dna_ids MUST also appear in the prompt as @ExactDNAName (e.g. "@Zohar walks…") — never "Zohar\'s" or "the man on the left" as a substitute. IMPORTANT: different models accept different numbers and durations of inputs — call list_models type="elements" and read elements_max_images / elements_max_videos / elements_max_audio plus min_video_duration / max_video_duration before generating. For text-only → video use generate_video instead. For animating a single still image use generate_video_from_image. Returns the final video URL when complete.',
1561
+ 'Generate a video from reference elements (images, videos, and/or audio) + a text prompt. SPECIAL MODEL EXCEPTION: seedance-2-5-multilingual accepts an ordinary prompt in any language with the ORIGINAL target-language dialogue in double quotes, including Hebrew script. Use this tool even without references for that model, duration 8 or 12, resolution 480p/720p/1080p (no Draft), up to 3 still references or Visual DNAs, no video/audio references. It performs translation, distinct designed voices and segmented lip sync internally; never transliterate or pre-translate its quoted dialogue. The following regular Seedance instructions apply only to other model IDs. Use when the user wants to animate specific uploaded/referenced assets — e.g. "animate this product", "put these 3 characters into a scene". PRIMARY ROUTE FOR A DNA-ANCHORED MULTI-SHOT FILM: one call can carry the whole sequence — seedance-2-5 takes 4-30s, up to 30 shots and 20 Visual DNAs in a SINGLE generation (seedance-2: 4-15s, 9 DNAs) — instead of a stack of separate clips. DIALOGUE IS PERFORMED NATIVELY: quoted dialogue in the prompt comes back as synced voices with lip movement, room tone and the SFX named in the AUDIO block — never route scene dialogue to generate_speech or generate_lipsync. Write dialogue in ENGLISH or Latin transliteration of Hebrew ("shalom"), never Hebrew script — Seedance does not speak Hebrew; prefer Gemini Omni Flash 1.1 or Gemini Omni 1 for native Hebrew. COST: resolution is a multiplier. When list_models publishes `video_input_credit` and this call carries videos, charge that rate against nominal input seconds + nominal output seconds; MP4 padding within 0.15s of an integer snaps to that integer and larger fractions round up. Otherwise use the normal output-second rate. PROMPT CONTRACT (Seedance / Elements): Locked Intro only — Total line, then [GLOBAL LOOK] / [CAST] / [LOCATION] / SHOT N. Do NOT write SCENE CONTEXT / OPTICS / ACTION department packs. Every Visual DNA in visual_dna_ids MUST also appear in the prompt as @ExactDNAName (e.g. "@Zohar walks…") — never "Zohar\'s" or "the man on the left" as a substitute. IMPORTANT: different models accept different numbers and durations of inputs — call list_models type="elements" and read elements_max_images / elements_max_videos / elements_max_audio plus min_video_duration / max_video_duration before generating. For text-only → video use generate_video instead. For animating a single still image use generate_video_from_image. Returns the final video URL when complete.' + MORPHIOUS_GUIDE,
1553
1562
  {
1554
- prompt: z.string().describe('For seedance-2-5-multilingual: ordinary scene description with original target-language dialogue in double quotes; no manual translation or transliteration. For other models, Locked Intro prompt (Seedance/Elements): Total line, [GLOBAL LOOK], [CAST] with @ExactDNAName for every visual_dna_ids entry, [LOCATION], then SHOT N. Not SCENE CONTEXT/OPTICS/ACTION packs. Never substitute "the left man" or "Zohar\'s" for @Name. EVERY attached reference must also be tagged by its 1-based array position — `@Image 1`/`@Image 2` (reference_images), `@Video 1` (reference_videos), `@Audio 1` (reference_audio_urls) — and its job stated ("@Image 1 defines the character\'s face and wardrobe", "@Video 1 defines the camera move"). An untagged attachment is ignored by the engine even though it was uploaded and billed.'),
1563
+ prompt: z.string().optional().describe('Optional ONLY for kolbo-morphious-* and higgsfield-genjutsu-* models (the server analyzes the media). For seedance-2-5-multilingual: ordinary scene description with original target-language dialogue in double quotes; no manual translation or transliteration. For other models, Locked Intro prompt (Seedance/Elements): Total line, [GLOBAL LOOK], [CAST] with @ExactDNAName for every visual_dna_ids entry, [LOCATION], then SHOT N. Not SCENE CONTEXT/OPTICS/ACTION packs. Never substitute "the left man" or "Zohar\'s" for @Name. EVERY attached reference must also be tagged by its 1-based array position — `@Image 1`/`@Image 2` (reference_images), `@Video 1` (reference_videos), `@Audio 1` (reference_audio_urls) — and its job stated ("@Image 1 defines the character\'s face and wardrobe", "@Video 1 defines the camera move"). An untagged attachment is ignored by the engine even though it was uploaded and billed.'),
1555
1564
  model: z.string().optional().describe('Model identifier. If the user already named a family (Grok / Kling / Veo / Seedance / …), pass THAT family — never default to Seedance because Elements often uses it. Use list_models type="elements" for exact ids and elements_max_* caps. Do NOT omit (omitting = Smart Select).'),
1556
1565
  reference_images: z.array(z.string()).optional().describe('Array of image references (product shots, character references, etc.). Tag each one in the prompt text by its 1-based position here — item 1 is `@Image 1`, item 2 is `@Image 2` — or the engine ignores it. Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_images` URLs from list_models for the chosen model — exceeding it is a deterministic 400.**'),
1557
1566
  reference_videos: z.array(z.string()).optional().describe('Array of reference videos for models that accept video inputs. Tag each one in the prompt text by its 1-based position here — `@Video 1`, `@Video 2` — stating what it defines (motion, camera move, pacing), or the engine ignores it. Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_videos` URLs and keep every clip within `min_video_duration`-`max_video_duration` from list_models.** If `video_input_credit` is present, every attached video contributes its nominal duration to combined-second billing: encoder padding within 0.15s of an integer snaps to it; larger fractions round up.'),
@@ -1577,14 +1586,14 @@ function registerGenerateTools(server, client, options = {}) {
1577
1586
  project_id: projectIdField,
1578
1587
  session_id: sessionIdField
1579
1588
  },
1580
- async ({ prompt, model, reference_images, reference_videos, reference_audio_urls, audio_url, files, duration, aspect_ratio, motion, preset_id, enhance_prompt = false, visual_dna_ids, resolution, draft, sound_enabled, keyframes, multi_shots, multi_shot_count, session_name, project_id, session_id }) => {
1589
+ async ({ prompt = '', model, reference_images, reference_videos, reference_audio_urls, audio_url, files, duration, aspect_ratio, motion, preset_id, enhance_prompt = false, visual_dna_ids, resolution, draft, sound_enabled, keyframes, multi_shots, multi_shot_count, session_name, project_id, session_id }) => {
1581
1590
  validateVideoPrompt({ prompt, duration, aspect_ratio, multi_shots, multi_shot_count });
1582
1591
  if (draft === true) resolution = '480p-draft';
1583
1592
  else if (draft === false && resolution?.endsWith('-draft')) resolution = resolution.slice(0, -6);
1584
1593
  model = await canonicalModelId(client, model, 'elements'); // lenient id resolution ("z-image" → "z-image/turbo")
1585
1594
  aspect_ratio = await resolveCatalogAspectRatio(client, model, aspect_ratio, 'elements');
1586
1595
  validateVideoPrompt({ prompt, duration, aspect_ratio, multi_shots, multi_shot_count });
1587
- if (!prompt) throw new Error('prompt is required');
1596
+ if (!prompt && !PROMPTLESS_NATIVE.has(model)) throw new Error('prompt is required');
1588
1597
 
1589
1598
  // Elements is the one tool that takes all three modalities, and either a
1590
1599
  // URL or a local path for any of them. Bucket every input by the kind the
@@ -1970,10 +1979,10 @@ function registerGenerateTools(server, client, options = {}) {
1970
1979
  // ─── generate_video_from_video ─────────────────────────────
1971
1980
  server.tool(
1972
1981
  'generate_video_from_video',
1973
- 'Restyle / transform an existing video (video-to-video). Use for style transfer, scene restyling, subject swap, motion transfer, character replacement, or burning in styled subtitles (VEED Subtitles). Source video can be a URL or absolute local path. `prompt` is OPTIONAL: most models need it, but prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control) ignore it. For VEED Subtitles, pass a `preset` style and optional `source_language` / `translation_language` instead of a prompt. IMPORTANT: different models support different extra inputs — call list_models type="video_to_video" and read max_images / max_videos / max_elements on the chosen model before generating. Pass reference_images for models with max_images > 0 (e.g. Kling O1/O3, Aleph, WAN VACE), reference_videos for models with max_videos > 1 (e.g. WAN 2.6 reference-to-video accepts up to 3), and elements for models with max_elements > 0. REPAIR / RETIME (not restyling, no prompt needed — these keep the footage and fix or retime it): model "topaz/deblur/video" removes lens, motion and compression blur; "topaz/colorize/video" colorizes black-and-white footage; "topaz/interpolate/video" is SLOW MOTION and frame-rate conversion (see slowdown_factor / target_fps); "topaz/sdr-to-hdr/video" masters SDR footage to HDR (see output_format). Reach for these when the user says blurry, shaky-detail, black-and-white, slow motion, smoother frame rate, or HDR — a restyle model would repaint the video instead of repairing it. For animating a still image use generate_video_from_image instead. For text-only → video use generate_video.',
1982
+ 'Restyle / transform an existing video (video-to-video). Use for style transfer, scene restyling, subject swap, motion transfer, character replacement, or burning in styled subtitles (VEED Subtitles). Source video can be a URL or absolute local path. `prompt` is OPTIONAL: most models need it, but prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control) ignore it. For VEED Subtitles, pass a `preset` style and optional `source_language` / `translation_language` instead of a prompt. IMPORTANT: different models support different extra inputs — call list_models type="video_to_video" and read max_images / max_videos / max_elements on the chosen model before generating. Pass reference_images for models with max_images > 0 (e.g. Kling O1/O3, Aleph, WAN VACE), reference_videos for models with max_videos > 1 (e.g. WAN 2.6 reference-to-video accepts up to 3), and elements for models with max_elements > 0. REPAIR / RETIME (not restyling, no prompt needed — these keep the footage and fix or retime it): model "topaz/deblur/video" removes lens, motion and compression blur; "topaz/colorize/video" colorizes black-and-white footage; "topaz/interpolate/video" is SLOW MOTION and frame-rate conversion (see slowdown_factor / target_fps); "topaz/sdr-to-hdr/video" masters SDR footage to HDR (see output_format). Reach for these when the user says blurry, shaky-detail, black-and-white, slow motion, smoother frame rate, or HDR — a restyle model would repaint the video instead of repairing it. For animating a still image use generate_video_from_image instead. For text-only → video use generate_video.' + MORPHIOUS_GUIDE,
1974
1983
  {
1975
1984
  source_video: z.string().describe('URL or absolute local path to the primary source video to restyle. **Source duration must fall within `min_video_duration`-`max_video_duration` from list_models for the chosen model** — videos outside that range are rejected (or silently truncated by some upstream providers). For models that use reference_videos as their primary input (e.g. WAN 2.6 reference-to-video), pass the first reference video here and also include it in reference_videos.'),
1976
- prompt: z.string().optional().describe('Text description of the desired restyle / transformation. Required by most video-to-video models; omit for prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control).'),
1985
+ prompt: z.string().optional().describe('Text description of the desired restyle / transformation. Required by most video-to-video models; omit for prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control, kolbo-morphious-*, higgsfield-genjutsu-motion-transfer, higgsfield-genjutsu-object-swap).'),
1977
1986
  model: z.string().optional().describe('Model identifier. Use list_models type="video_to_video" to see options and check max_images / max_videos / max_elements / max_video_duration per model. Pick a SPECIFIC model — do NOT omit (omitting = Smart Select auto-pick, which we avoid); call list_models for this type and choose the model that best fits the user\'s intent.'),
1978
1987
  aspect_ratio: z.string().optional().describe(aspectRatioDescribe() + ' Default: matches source.'),
1979
1988
  duration: z.number().optional().describe('Output duration in seconds. Must be in `supported_durations` from list_models, OR within `min_output_duration`-`max_output_duration`. Default: matches source'),
@@ -1,6 +1,6 @@
1
1
  // Validate only explicit, machine-readable declarations. Never infer creative
2
2
  // intent from adjectives, require a template, or rewrite a user's prompt.
3
- function validateVideoPrompt({ prompt, duration, aspect_ratio, multi_shots, multi_shot_count }) {
3
+ function validateVideoPrompt({ prompt = '', duration, aspect_ratio, multi_shots, multi_shot_count }) {
4
4
  const totals = [...prompt.matchAll(/^\s*Total:\s*(\d+(?:\.\d+)?)s\s*\/\s*(\d+)\s*shots?\s*\/\s*(\d+:\d+)\s*$/gim)]
5
5
  .map((m) => ({ duration: Number(m[1]), count: Number(m[2]), aspect: m[3] }));
6
6
  if (!totals.length) return; // Existing free-form clients remain supported.