@kolbo/mcp 1.99.0 → 1.100.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kolbo/mcp",
3
- "version": "1.99.0",
3
+ "version": "1.100.0",
4
4
  "description": "Kolbo AI MCP Server - Generate images, videos, music, speech, and sound effects from Claude Code",
5
5
  "main": "src/index.js",
6
6
  "bin": {
@@ -1,6 +1,6 @@
1
1
  # AUTO-GENERATED — do not edit
2
2
 
3
- This tree is mirrored from kolbo-code@650ecac, the single source of truth.
3
+ This tree is mirrored from kolbo-code@e4764c5, the single source of truth.
4
4
  Canonical source: packages/opencode/skills/kolbo/
5
5
  Distribution: .github/workflows/sync-skill-to-plugin.yml
6
6
 
package/skill/SKILL.md CHANGED
@@ -165,7 +165,8 @@ Font tools (when exposed by the installed MCP): `list_fonts`, `get_font`, `uploa
165
165
  | `list_project_assets` / `link_project_asset` / `unlink_project_asset` / `update_project_asset` | Project CAST roster: the Visual DNAs and moodboards tagged onto a project (`@Name` / `#Name`). `update_project_asset` writes each tagged DNA's identity description and/or its project-scoped purpose note. Never unlink+relink to edit. |
166
166
  | `create_moodboard` / `update_moodboard` / `delete_moodboard` | Moodboards from image URLs → AI master style prompt → pass `moodboard_id` to generation tools. Edit with `update_moodboard`; never delete+recreate. |
167
167
  | `clone_voice` / `import_elevenlabs_voice` / `delete_voice` | Custom voices (clone CHARGES CREDITS — confirm first; new voices show in `list_voices`) |
168
- | `trim_video` | Frame-accurate trim of a Kolbo-hosted video (tool waits and returns the URL). `edit_video` also gained `remove_background`. |
168
+ | `trim_video` | Frame-accurate trim of a Kolbo-hosted video (tool waits and returns the URL). `edit_video` also gained `remove_background`. |
169
+ | `edit_video` input previews | Edit widgets retain the source video, optional mask video, face image, and audio references through generation and completion. Local edit files use their uploaded CDN URLs for previews. Click a video thumbnail to inspect the source. |
169
170
  | `create_doc` / `list_docs` / `get_doc` / `update_doc` / `share_doc` / `delete_doc` | AI Docs (Magic Pad): YOU author full HTML documents (plans, briefs, scripts, research) saved into the user's project, editable in the Kolbo app. `share_doc` returns a public link. `update_doc` content replaces the WHOLE doc — `get_doc` first. |
170
171
  | `chat_send_message` / `chat_list_conversations` / `chat_get_messages` | Kolbo chat with optional `media_urls` (up to 10 per call) and `thinking_level` from `list_models` type `text` thinkingLevels; omitted/invalid levels use the resolved model default, safeguards and legacy `deep_think` take precedence |
171
172
  | `create_review_asset` / `add_review_version` / `set_review_status` / `create_review_comment` / `reply_review_comment` / `resolve_review_comment` / `unresolve_review_comment` / `create_review_collection` / `create_review_share_link` / `revoke_review_share_link` / `get_review_storage_usage` (+ list/get/update/delete siblings) | **Kolbo Review** — Frame.io-style client review: asset = media + appended versions (new cut = `add_review_version`, never delete+recreate), timecoded comments per version, approve/request-changes status, guest share links (no Kolbo account; comment-only unless `canSetStatus`). 5GB review storage cap. See `workflows/review-collections.md`. |
@@ -33,6 +33,8 @@ const READ_ONLY = [
33
33
  'get_music_track_related', 'get_music_track_lyrics',
34
34
  'search_stock_media', 'get_stock_sources', 'get_stock_categories',
35
35
  'get_stock_collections', 'get_stock_asset', 'analyze_script_for_stock',
36
+ // Free (no credits, no DB write) — returns a scratch composition plan for the caller to edit.
37
+ 'create_music_composition_plan',
36
38
  ];
37
39
 
38
40
  const OPEN_WORLD_READ_ONLY = [
@@ -85,6 +87,9 @@ const DESTRUCTIVE_WRITE = [
85
87
  'generate_image', 'generate_image_edit', 'generate_creative_director',
86
88
  'generate_video', 'generate_video_from_image', 'generate_music',
87
89
  'generate_speech', 'generate_sound', 'cancel_generation',
90
+ // Both spend credits — reference-audio upload is billed by ElevenLabs like a generation,
91
+ // section edit creates a new billed track.
92
+ 'create_music_reference_audio', 'edit_music_section',
88
93
  'generate_elements', 'generate_first_last_frame', 'generate_lipsync',
89
94
  'generate_video_from_video', 'transcribe_audio', 'generate_3d',
90
95
  'edit_image', 'edit_video', 'trim_video', 'clone_voice',
@@ -57,6 +57,27 @@ const CINEMATIC_SCHEMA = z.object({
57
57
  // alongside the prompt (generate_video_from_image: `{ image_url, prompt }` — the
58
58
  // image is what varies, and that is the whole point). Either way the widget
59
59
  // captions each tile with the item's prompt, so that is the label we carry.
60
+ // ElevenLabs Music v2.5 composition-plan chunk — one entry is EITHER a generation chunk
61
+ // (text/duration_ms/positive_styles, optionally conditioning_ref to steer its style off a
62
+ // stored song) OR an audio-reference chunk (song_id/range, splices that slice in unchanged).
63
+ // All fields optional at the schema level; the backend validates the real constraints
64
+ // (3-120s per generation chunk, <=30s conditioning reference, <=30 chunks per plan).
65
+ const MUSIC_TIME_RANGE_SCHEMA = z.object({
66
+ start_ms: z.number().int().min(0),
67
+ end_ms: z.number().int().min(0),
68
+ }).describe('{ start_ms, end_ms } — a slice of a stored song, in milliseconds.');
69
+ const MUSIC_CHUNK_SCHEMA = z.object({
70
+ text: z.string().optional().describe('Generation chunk only: section name in [brackets], lyric lines, {inline directions}. E.g. "[Chorus]\\nWe rise tonight".'),
71
+ duration_ms: z.number().int().min(3000).max(120000).optional().describe('Generation chunk only: length in ms, 3000-120000.'),
72
+ positive_styles: z.array(z.string()).optional().describe('Generation chunk only: styles/directions to include (max 50). The first chunk\'s styles set the whole song\'s tone.'),
73
+ negative_styles: z.array(z.string()).optional().describe('Generation chunk only: styles/directions to avoid.'),
74
+ context_adherence: z.enum(['low', 'medium', 'high']).optional().describe('Generation chunk only: how closely it follows neighboring chunks. Default high.'),
75
+ conditioning_ref: z.object({ song_id: z.string(), range: MUSIC_TIME_RANGE_SCHEMA }).optional().describe('Generation chunk only: condition this chunk\'s STYLE on <=30s of a stored song (does not splice audio in — it steers a fresh generation).'),
76
+ condition_strength: z.enum(['low', 'medium', 'high', 'xhigh']).optional().describe('Generation chunk only: how strongly it follows conditioning_ref. Default medium.'),
77
+ song_id: z.string().optional().describe('Audio-reference chunk only: id of a stored song (from a prior generate_music result\'s song_id, or create_music_reference_audio) to splice in unchanged.'),
78
+ range: MUSIC_TIME_RANGE_SCHEMA.optional().describe('Audio-reference chunk only: the slice of song_id to insert unchanged.'),
79
+ });
80
+
60
81
  const MAX_BATCH_PROMPTS = 8;
61
82
  const MAX_STATUS_IDS = 20;
62
83
  async function submitBatch(rawItems, submitOne) {
@@ -797,7 +818,7 @@ function registerGenerateTools(server, client, options = {}) {
797
818
  'generate_music',
798
819
  'Generate music from a text description using Kolbo AI. Supports instrumental mode, custom lyrics, style direction, vocal gender, negative tags, song length, and Suno fine-controls (style weight, weirdness, audio weight, persona/singing voice). Default model is Suno. Some controls are Suno-only; the engine ignores controls that do not apply to the chosen model. Returns the final audio URL when complete.',
799
820
  {
800
- prompt: z.string().describe('Text description of the music to generate (e.g., "upbeat electronic dance track with synthesizers")'),
821
+ prompt: z.string().optional().describe('Text description of the music to generate (e.g., "upbeat electronic dance track with synthesizers"). Required unless composition_plan is passed (ElevenLabs Music v2.5 structured mode, which cannot be combined with prompt).'),
801
822
  model: z.string().optional().describe('Model identifier. Use list_models type="music_gen" to see options. Omit for Suno (default).'),
802
823
  style: z.string().optional().describe('Music style / genre (e.g., "pop", "rock", "lo-fi", "electronic", "jazz")'),
803
824
  title: z.string().optional().describe('Song title. If omitted, one is generated.'),
@@ -816,16 +837,22 @@ function registerGenerateTools(server, client, options = {}) {
816
837
  use_composition_plan: z.boolean().optional().describe('Suno: enable structured composition planning (verse/chorus structure).'),
817
838
  singing_dna_id: z.string().optional().describe('Visual DNA character id whose singing voice to use (must be owned by the caller).'),
818
839
  singing_voice_id: z.string().optional().describe('Custom cloned singing-voice id (must be owned by the caller).'),
840
+ // ── ElevenLabs Music v2.5 structured mode (mutually exclusive with `prompt`) ──
841
+ composition_plan: z.array(MUSIC_CHUNK_SCHEMA).optional().describe('ElevenLabs Music v2.5 ONLY, model must resolve to the ElevenLabs music model: an ordered list of chunks instead of a prompt. A GENERATION chunk has text/duration_ms/positive_styles (optionally conditioning_ref+condition_strength to steer it off a stored song). An AUDIO-REFERENCE chunk has song_id/range and splices that stored slice in unchanged (used to keep part of an existing track — see edit_music_section for the common case of regenerating one section). Get a starter plan from create_music_composition_plan, or build one from a prior generate_music result\'s `song_id`. Cannot be combined with `prompt`.'),
842
+ seed: z.number().optional().describe('ElevenLabs Music v2.5: random seed for reproducibility. Only valid together with composition_plan, never with prompt.'),
819
843
  project_id: projectIdField,
820
844
  session_id: sessionIdField
821
845
  },
822
- async ({ prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags, duration_seconds, enhance_prompt = false, preset_id, style_weight, weirdness, audio_weight, persona_id, use_composition_plan, singing_dna_id, singing_voice_id, project_id, session_id }) => {
846
+ async ({ prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags, duration_seconds, enhance_prompt = false, preset_id, style_weight, weirdness, audio_weight, persona_id, use_composition_plan, singing_dna_id, singing_voice_id, composition_plan, seed, project_id, session_id }) => {
847
+ if (!prompt && !composition_plan) throw new Error('prompt is required (or pass composition_plan for ElevenLabs Music v2.5)');
823
848
  model = await canonicalModelId(client, model, 'music_gen'); // lenient id resolution ("z-image" → "z-image/turbo")
824
849
  const gen = await client.post('/v1/generate/music', {
825
850
  prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags,
826
851
  duration_seconds, enhance_prompt, preset_id,
827
852
  style_weight, weirdness, audio_weight, persona_id, use_composition_plan,
828
- singing_dna_id, singing_voice_id, project_id, session_id
853
+ singing_dna_id, singing_voice_id,
854
+ ...(composition_plan ? { composition_plan: { chunks: composition_plan }, seed } : {}),
855
+ project_id, session_id
829
856
  });
830
857
 
831
858
  if (returnsImmediately()) return submittedResult({
@@ -855,11 +882,99 @@ function registerGenerateTools(server, client, options = {}) {
855
882
  playback_urls: result.result.playback_urls,
856
883
  title: result.result.title,
857
884
  duration: result.result.duration,
858
- lyrics: result.result.lyrics
885
+ lyrics: result.result.lyrics,
886
+ // ElevenLabs Music v2.5 only — reuse in a later composition_plan / edit_music_section.
887
+ ...(result.result.song_id ? { song_id: result.result.song_id } : {}),
888
+ }, null, 2));
889
+ }
890
+ );
891
+
892
+ // ─── create_music_composition_plan ─────────────────────────
893
+ // FREE (no credits, no generation created) — ElevenLabs Music v2.5's POST /v1/music/plan.
894
+ server.tool(
895
+ 'create_music_composition_plan',
896
+ 'Generate a starter ElevenLabs Music v2.5 structured composition plan (an ordered list of chunks with section text, styles, and durations) from a plain-text prompt. FREE — creates no generation and charges no credits. Edit the returned chunks and pass them as `composition_plan` to generate_music to actually create the track.',
897
+ {
898
+ prompt: z.string().describe('Text description of the song to plan (e.g. "an upbeat pop song with verse and chorus about summer").'),
899
+ duration_seconds: z.number().optional().describe('Target total length in seconds, 3-600. Omit to let the model choose.'),
900
+ project_id: projectIdField,
901
+ session_id: sessionIdField,
902
+ },
903
+ async ({ prompt, duration_seconds, project_id, session_id }) => {
904
+ const res = await client.post('/v1/generate/music/composition-plan', { prompt, duration_seconds, project_id, session_id });
905
+ return { content: [{ type: 'text', text: JSON.stringify(res, null, 2) }] };
906
+ }
907
+ );
908
+
909
+ // ─── create_music_reference_audio ───────────────────────────
910
+ // BILLED at the same rate as a song generation of the clip's length (ElevenLabs charges
911
+ // the upload itself, even half-price if it flags the clip for copyright) — this is NOT a
912
+ // free utility call, unlike create_music_composition_plan above.
913
+ server.tool(
914
+ 'create_music_reference_audio',
915
+ 'Upload an existing audio clip you already have in Kolbo so it can be used as an audio reference in a later generate_music composition_plan (as a song_id in a conditioning_ref, to steer style, or in an audio-reference chunk, to splice a slice in unchanged) or as the source for edit_music_section. BILLED the same as generating a track of the clip\'s length — ElevenLabs charges for this upload itself. LOCAL FILE? Call upload_media first and pass the returned media id here.',
916
+ {
917
+ media_id: z.string().describe('A media-library item id for an audio file you own — from upload_media, or from list_media (mediaType audio).'),
918
+ project_id: projectIdField,
919
+ session_id: sessionIdField,
920
+ },
921
+ async ({ media_id, project_id, session_id }) => {
922
+ const res = await client.post('/v1/generate/music/reference-audio', { media_id, project_id, session_id });
923
+ return { content: [{ type: 'text', text: JSON.stringify(res, null, 2) }] };
924
+ }
925
+ );
926
+
927
+ // ─── edit_music_section ─────────────────────────────────────
928
+ // ElevenLabs Music v2.5 inpainting — regenerate one time range of an existing track,
929
+ // keeping the rest unchanged. BILLED like an ordinary generation (creates a new track).
930
+ server.tool(
931
+ 'edit_music_section',
932
+ 'Regenerate one section (time range) of an ElevenLabs Music v2.5 track you already generated, keeping the rest of the song unchanged — e.g. "redo the outro with different lyrics". Only works on tracks made with the ElevenLabs Music model (check the earlier generate_music result for a `song_id`; a track without one cannot be edited this way). Creates a NEW generation in the same session; the source track is untouched. Returns a generation_id to poll like any other generate_music call.',
933
+ {
934
+ source_generation_id: z.string().describe('The generation_id of the ElevenLabs Music v2.5 track to edit (must have returned a song_id).'),
935
+ start_ms: z.number().int().min(0).describe('Start of the region to regenerate, in milliseconds.'),
936
+ end_ms: z.number().int().min(0).describe('End of the region to regenerate, in milliseconds. (end_ms - start_ms) must be 3000-120000ms.'),
937
+ text: z.string().optional().describe('New section text — [Section Name], lyric lines, {inline directions}. Omit to keep it instrumental/unlabeled.'),
938
+ positive_styles: z.array(z.string()).optional().describe('Styles/directions for the regenerated section.'),
939
+ negative_styles: z.array(z.string()).optional().describe('Styles/directions to avoid in the regenerated section.'),
940
+ context_adherence: z.enum(['low', 'medium', 'high']).optional().describe('How closely the new section follows the kept audio around it. Default high.'),
941
+ project_id: projectIdField,
942
+ session_id: sessionIdField,
943
+ },
944
+ async ({ source_generation_id, start_ms, end_ms, text, positive_styles, negative_styles, context_adherence, project_id, session_id }) => {
945
+ const gen = await client.post('/v1/generate/music/edit-section', {
946
+ source_generation_id, start_ms, end_ms, text,
947
+ positive_styles, negative_styles, context_adherence,
948
+ project_id, session_id,
949
+ });
950
+
951
+ if (returnsImmediately()) return submittedResult({
952
+ tool: 'edit_music_section', kind: 'audio', gen, client, model: 'ElevenLabs Music', prompt: text || '(section edit)',
953
+ });
954
+
955
+ const poll = await pollOrTimedOut(client, gen.generation_id, { interval: (gen.poll_interval_hint || 8) * 1000, timeout: 150000 });
956
+ if (poll.timedOut) return poll.timedOut;
957
+ const result = poll.result;
958
+
959
+ return uiCompleted({
960
+ tool: 'edit_music_section', kind: 'audio', gen, client, model: 'ElevenLabs Music', prompt: text || '(section edit)',
961
+ urls: result.result.urls,
962
+ playback_urls: result.result.playback_urls,
963
+ title: result.result.title,
964
+ duration: result.result.duration,
965
+ credits_used: creditFields(result).credits_used,
966
+ }, JSON.stringify({
967
+ ...creditFields(result),
968
+ session_id: gen.session_id,
969
+ urls: result.result.urls,
970
+ playback_urls: result.result.playback_urls,
971
+ duration: result.result.duration,
972
+ ...(result.result.song_id ? { song_id: result.result.song_id } : {}),
859
973
  }, null, 2));
860
974
  }
861
975
  );
862
976
 
977
+
863
978
  // ─── music import / extend / cover ─────────────────────────
864
979
  // Three tools over the SDK's /v1/generate/music/{import,extend,cover}. The web app has
865
980
  // had upload-extend and upload-cover for a long time; agents could not reach either