@kolbo/mcp 1.99.1 → 1.100.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/toolAnnotations.js +5 -0
- package/src/tools/generate.js +119 -4
package/package.json
CHANGED
package/src/toolAnnotations.js
CHANGED
|
@@ -33,6 +33,8 @@ const READ_ONLY = [
|
|
|
33
33
|
'get_music_track_related', 'get_music_track_lyrics',
|
|
34
34
|
'search_stock_media', 'get_stock_sources', 'get_stock_categories',
|
|
35
35
|
'get_stock_collections', 'get_stock_asset', 'analyze_script_for_stock',
|
|
36
|
+
// Free (no credits, no DB write) — returns a scratch composition plan for the caller to edit.
|
|
37
|
+
'create_music_composition_plan',
|
|
36
38
|
];
|
|
37
39
|
|
|
38
40
|
const OPEN_WORLD_READ_ONLY = [
|
|
@@ -85,6 +87,9 @@ const DESTRUCTIVE_WRITE = [
|
|
|
85
87
|
'generate_image', 'generate_image_edit', 'generate_creative_director',
|
|
86
88
|
'generate_video', 'generate_video_from_image', 'generate_music',
|
|
87
89
|
'generate_speech', 'generate_sound', 'cancel_generation',
|
|
90
|
+
// Both spend credits — reference-audio upload is billed by ElevenLabs like a generation,
|
|
91
|
+
// section edit creates a new billed track.
|
|
92
|
+
'create_music_reference_audio', 'edit_music_section',
|
|
88
93
|
'generate_elements', 'generate_first_last_frame', 'generate_lipsync',
|
|
89
94
|
'generate_video_from_video', 'transcribe_audio', 'generate_3d',
|
|
90
95
|
'edit_image', 'edit_video', 'trim_video', 'clone_voice',
|
package/src/tools/generate.js
CHANGED
|
@@ -57,6 +57,27 @@ const CINEMATIC_SCHEMA = z.object({
|
|
|
57
57
|
// alongside the prompt (generate_video_from_image: `{ image_url, prompt }` — the
|
|
58
58
|
// image is what varies, and that is the whole point). Either way the widget
|
|
59
59
|
// captions each tile with the item's prompt, so that is the label we carry.
|
|
60
|
+
// ElevenLabs Music v2.5 composition-plan chunk — one entry is EITHER a generation chunk
|
|
61
|
+
// (text/duration_ms/positive_styles, optionally conditioning_ref to steer its style off a
|
|
62
|
+
// stored song) OR an audio-reference chunk (song_id/range, splices that slice in unchanged).
|
|
63
|
+
// All fields optional at the schema level; the backend validates the real constraints
|
|
64
|
+
// (3-120s per generation chunk, <=30s conditioning reference, <=30 chunks per plan).
|
|
65
|
+
const MUSIC_TIME_RANGE_SCHEMA = z.object({
|
|
66
|
+
start_ms: z.number().int().min(0),
|
|
67
|
+
end_ms: z.number().int().min(0),
|
|
68
|
+
}).describe('{ start_ms, end_ms } — a slice of a stored song, in milliseconds.');
|
|
69
|
+
const MUSIC_CHUNK_SCHEMA = z.object({
|
|
70
|
+
text: z.string().optional().describe('Generation chunk only: section name in [brackets], lyric lines, {inline directions}. E.g. "[Chorus]\\nWe rise tonight".'),
|
|
71
|
+
duration_ms: z.number().int().min(3000).max(120000).optional().describe('Generation chunk only: length in ms, 3000-120000.'),
|
|
72
|
+
positive_styles: z.array(z.string()).optional().describe('Generation chunk only: styles/directions to include (max 50). The first chunk\'s styles set the whole song\'s tone.'),
|
|
73
|
+
negative_styles: z.array(z.string()).optional().describe('Generation chunk only: styles/directions to avoid.'),
|
|
74
|
+
context_adherence: z.enum(['low', 'medium', 'high']).optional().describe('Generation chunk only: how closely it follows neighboring chunks. Default high.'),
|
|
75
|
+
conditioning_ref: z.object({ song_id: z.string(), range: MUSIC_TIME_RANGE_SCHEMA }).optional().describe('Generation chunk only: condition this chunk\'s STYLE on <=30s of a stored song (does not splice audio in — it steers a fresh generation).'),
|
|
76
|
+
condition_strength: z.enum(['low', 'medium', 'high', 'xhigh']).optional().describe('Generation chunk only: how strongly it follows conditioning_ref. Default medium.'),
|
|
77
|
+
song_id: z.string().optional().describe('Audio-reference chunk only: id of a stored song (from a prior generate_music result\'s song_id, or create_music_reference_audio) to splice in unchanged.'),
|
|
78
|
+
range: MUSIC_TIME_RANGE_SCHEMA.optional().describe('Audio-reference chunk only: the slice of song_id to insert unchanged.'),
|
|
79
|
+
});
|
|
80
|
+
|
|
60
81
|
const MAX_BATCH_PROMPTS = 8;
|
|
61
82
|
const MAX_STATUS_IDS = 20;
|
|
62
83
|
async function submitBatch(rawItems, submitOne) {
|
|
@@ -797,7 +818,7 @@ function registerGenerateTools(server, client, options = {}) {
|
|
|
797
818
|
'generate_music',
|
|
798
819
|
'Generate music from a text description using Kolbo AI. Supports instrumental mode, custom lyrics, style direction, vocal gender, negative tags, song length, and Suno fine-controls (style weight, weirdness, audio weight, persona/singing voice). Default model is Suno. Some controls are Suno-only; the engine ignores controls that do not apply to the chosen model. Returns the final audio URL when complete.',
|
|
799
820
|
{
|
|
800
|
-
prompt: z.string().describe('Text description of the music to generate (e.g., "upbeat electronic dance track with synthesizers")'),
|
|
821
|
+
prompt: z.string().optional().describe('Text description of the music to generate (e.g., "upbeat electronic dance track with synthesizers"). Required unless composition_plan is passed (ElevenLabs Music v2.5 structured mode, which cannot be combined with prompt).'),
|
|
801
822
|
model: z.string().optional().describe('Model identifier. Use list_models type="music_gen" to see options. Omit for Suno (default).'),
|
|
802
823
|
style: z.string().optional().describe('Music style / genre (e.g., "pop", "rock", "lo-fi", "electronic", "jazz")'),
|
|
803
824
|
title: z.string().optional().describe('Song title. If omitted, one is generated.'),
|
|
@@ -816,16 +837,22 @@ function registerGenerateTools(server, client, options = {}) {
|
|
|
816
837
|
use_composition_plan: z.boolean().optional().describe('Suno: enable structured composition planning (verse/chorus structure).'),
|
|
817
838
|
singing_dna_id: z.string().optional().describe('Visual DNA character id whose singing voice to use (must be owned by the caller).'),
|
|
818
839
|
singing_voice_id: z.string().optional().describe('Custom cloned singing-voice id (must be owned by the caller).'),
|
|
840
|
+
// ── ElevenLabs Music v2.5 structured mode (mutually exclusive with `prompt`) ──
|
|
841
|
+
composition_plan: z.array(MUSIC_CHUNK_SCHEMA).optional().describe('ElevenLabs Music v2.5 ONLY, model must resolve to the ElevenLabs music model: an ordered list of chunks instead of a prompt. A GENERATION chunk has text/duration_ms/positive_styles (optionally conditioning_ref+condition_strength to steer it off a stored song). An AUDIO-REFERENCE chunk has song_id/range and splices that stored slice in unchanged (used to keep part of an existing track — see edit_music_section for the common case of regenerating one section). Get a starter plan from create_music_composition_plan, or build one from a prior generate_music result\'s `song_id`. Cannot be combined with `prompt`.'),
|
|
842
|
+
seed: z.number().optional().describe('ElevenLabs Music v2.5: random seed for reproducibility. Only valid together with composition_plan, never with prompt.'),
|
|
819
843
|
project_id: projectIdField,
|
|
820
844
|
session_id: sessionIdField
|
|
821
845
|
},
|
|
822
|
-
async ({ prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags, duration_seconds, enhance_prompt = false, preset_id, style_weight, weirdness, audio_weight, persona_id, use_composition_plan, singing_dna_id, singing_voice_id, project_id, session_id }) => {
|
|
846
|
+
async ({ prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags, duration_seconds, enhance_prompt = false, preset_id, style_weight, weirdness, audio_weight, persona_id, use_composition_plan, singing_dna_id, singing_voice_id, composition_plan, seed, project_id, session_id }) => {
|
|
847
|
+
if (!prompt && !composition_plan) throw new Error('prompt is required (or pass composition_plan for ElevenLabs Music v2.5)');
|
|
823
848
|
model = await canonicalModelId(client, model, 'music_gen'); // lenient id resolution ("z-image" → "z-image/turbo")
|
|
824
849
|
const gen = await client.post('/v1/generate/music', {
|
|
825
850
|
prompt, model, style, title, instrumental, lyrics, vocal_gender, negative_tags,
|
|
826
851
|
duration_seconds, enhance_prompt, preset_id,
|
|
827
852
|
style_weight, weirdness, audio_weight, persona_id, use_composition_plan,
|
|
828
|
-
singing_dna_id, singing_voice_id,
|
|
853
|
+
singing_dna_id, singing_voice_id,
|
|
854
|
+
...(composition_plan ? { composition_plan: { chunks: composition_plan }, seed } : {}),
|
|
855
|
+
project_id, session_id
|
|
829
856
|
});
|
|
830
857
|
|
|
831
858
|
if (returnsImmediately()) return submittedResult({
|
|
@@ -855,11 +882,99 @@ function registerGenerateTools(server, client, options = {}) {
|
|
|
855
882
|
playback_urls: result.result.playback_urls,
|
|
856
883
|
title: result.result.title,
|
|
857
884
|
duration: result.result.duration,
|
|
858
|
-
lyrics: result.result.lyrics
|
|
885
|
+
lyrics: result.result.lyrics,
|
|
886
|
+
// ElevenLabs Music v2.5 only — reuse in a later composition_plan / edit_music_section.
|
|
887
|
+
...(result.result.song_id ? { song_id: result.result.song_id } : {}),
|
|
888
|
+
}, null, 2));
|
|
889
|
+
}
|
|
890
|
+
);
|
|
891
|
+
|
|
892
|
+
// ─── create_music_composition_plan ─────────────────────────
|
|
893
|
+
// FREE (no credits, no generation created) — ElevenLabs Music v2.5's POST /v1/music/plan.
|
|
894
|
+
server.tool(
|
|
895
|
+
'create_music_composition_plan',
|
|
896
|
+
'Generate a starter ElevenLabs Music v2.5 structured composition plan (an ordered list of chunks with section text, styles, and durations) from a plain-text prompt. FREE — creates no generation and charges no credits. Edit the returned chunks and pass them as `composition_plan` to generate_music to actually create the track.',
|
|
897
|
+
{
|
|
898
|
+
prompt: z.string().describe('Text description of the song to plan (e.g. "an upbeat pop song with verse and chorus about summer").'),
|
|
899
|
+
duration_seconds: z.number().optional().describe('Target total length in seconds, 3-600. Omit to let the model choose.'),
|
|
900
|
+
project_id: projectIdField,
|
|
901
|
+
session_id: sessionIdField,
|
|
902
|
+
},
|
|
903
|
+
async ({ prompt, duration_seconds, project_id, session_id }) => {
|
|
904
|
+
const res = await client.post('/v1/generate/music/composition-plan', { prompt, duration_seconds, project_id, session_id });
|
|
905
|
+
return { content: [{ type: 'text', text: JSON.stringify(res, null, 2) }] };
|
|
906
|
+
}
|
|
907
|
+
);
|
|
908
|
+
|
|
909
|
+
// ─── create_music_reference_audio ───────────────────────────
|
|
910
|
+
// BILLED at the same rate as a song generation of the clip's length (ElevenLabs charges
|
|
911
|
+
// the upload itself, even half-price if it flags the clip for copyright) — this is NOT a
|
|
912
|
+
// free utility call, unlike create_music_composition_plan above.
|
|
913
|
+
server.tool(
|
|
914
|
+
'create_music_reference_audio',
|
|
915
|
+
'Upload an existing audio clip you already have in Kolbo so it can be used as an audio reference in a later generate_music composition_plan (as a song_id in a conditioning_ref, to steer style, or in an audio-reference chunk, to splice a slice in unchanged) or as the source for edit_music_section. BILLED the same as generating a track of the clip\'s length — ElevenLabs charges for this upload itself. LOCAL FILE? Call upload_media first and pass the returned media id here.',
|
|
916
|
+
{
|
|
917
|
+
media_id: z.string().describe('A media-library item id for an audio file you own — from upload_media, or from list_media (mediaType audio).'),
|
|
918
|
+
project_id: projectIdField,
|
|
919
|
+
session_id: sessionIdField,
|
|
920
|
+
},
|
|
921
|
+
async ({ media_id, project_id, session_id }) => {
|
|
922
|
+
const res = await client.post('/v1/generate/music/reference-audio', { media_id, project_id, session_id });
|
|
923
|
+
return { content: [{ type: 'text', text: JSON.stringify(res, null, 2) }] };
|
|
924
|
+
}
|
|
925
|
+
);
|
|
926
|
+
|
|
927
|
+
// ─── edit_music_section ─────────────────────────────────────
|
|
928
|
+
// ElevenLabs Music v2.5 inpainting — regenerate one time range of an existing track,
|
|
929
|
+
// keeping the rest unchanged. BILLED like an ordinary generation (creates a new track).
|
|
930
|
+
server.tool(
|
|
931
|
+
'edit_music_section',
|
|
932
|
+
'Regenerate one section (time range) of an ElevenLabs Music v2.5 track you already generated, keeping the rest of the song unchanged — e.g. "redo the outro with different lyrics". Only works on tracks made with the ElevenLabs Music model (check the earlier generate_music result for a `song_id`; a track without one cannot be edited this way). Creates a NEW generation in the same session; the source track is untouched. Returns a generation_id to poll like any other generate_music call.',
|
|
933
|
+
{
|
|
934
|
+
source_generation_id: z.string().describe('The generation_id of the ElevenLabs Music v2.5 track to edit (must have returned a song_id).'),
|
|
935
|
+
start_ms: z.number().int().min(0).describe('Start of the region to regenerate, in milliseconds.'),
|
|
936
|
+
end_ms: z.number().int().min(0).describe('End of the region to regenerate, in milliseconds. (end_ms - start_ms) must be 3000-120000ms.'),
|
|
937
|
+
text: z.string().optional().describe('New section text — [Section Name], lyric lines, {inline directions}. Omit to keep it instrumental/unlabeled.'),
|
|
938
|
+
positive_styles: z.array(z.string()).optional().describe('Styles/directions for the regenerated section.'),
|
|
939
|
+
negative_styles: z.array(z.string()).optional().describe('Styles/directions to avoid in the regenerated section.'),
|
|
940
|
+
context_adherence: z.enum(['low', 'medium', 'high']).optional().describe('How closely the new section follows the kept audio around it. Default high.'),
|
|
941
|
+
project_id: projectIdField,
|
|
942
|
+
session_id: sessionIdField,
|
|
943
|
+
},
|
|
944
|
+
async ({ source_generation_id, start_ms, end_ms, text, positive_styles, negative_styles, context_adherence, project_id, session_id }) => {
|
|
945
|
+
const gen = await client.post('/v1/generate/music/edit-section', {
|
|
946
|
+
source_generation_id, start_ms, end_ms, text,
|
|
947
|
+
positive_styles, negative_styles, context_adherence,
|
|
948
|
+
project_id, session_id,
|
|
949
|
+
});
|
|
950
|
+
|
|
951
|
+
if (returnsImmediately()) return submittedResult({
|
|
952
|
+
tool: 'edit_music_section', kind: 'audio', gen, client, model: 'ElevenLabs Music', prompt: text || '(section edit)',
|
|
953
|
+
});
|
|
954
|
+
|
|
955
|
+
const poll = await pollOrTimedOut(client, gen.generation_id, { interval: (gen.poll_interval_hint || 8) * 1000, timeout: 150000 });
|
|
956
|
+
if (poll.timedOut) return poll.timedOut;
|
|
957
|
+
const result = poll.result;
|
|
958
|
+
|
|
959
|
+
return uiCompleted({
|
|
960
|
+
tool: 'edit_music_section', kind: 'audio', gen, client, model: 'ElevenLabs Music', prompt: text || '(section edit)',
|
|
961
|
+
urls: result.result.urls,
|
|
962
|
+
playback_urls: result.result.playback_urls,
|
|
963
|
+
title: result.result.title,
|
|
964
|
+
duration: result.result.duration,
|
|
965
|
+
credits_used: creditFields(result).credits_used,
|
|
966
|
+
}, JSON.stringify({
|
|
967
|
+
...creditFields(result),
|
|
968
|
+
session_id: gen.session_id,
|
|
969
|
+
urls: result.result.urls,
|
|
970
|
+
playback_urls: result.result.playback_urls,
|
|
971
|
+
duration: result.result.duration,
|
|
972
|
+
...(result.result.song_id ? { song_id: result.result.song_id } : {}),
|
|
859
973
|
}, null, 2));
|
|
860
974
|
}
|
|
861
975
|
);
|
|
862
976
|
|
|
977
|
+
|
|
863
978
|
// ─── music import / extend / cover ─────────────────────────
|
|
864
979
|
// Three tools over the SDK's /v1/generate/music/{import,extend,cover}. The web app has
|
|
865
980
|
// had upload-extend and upload-cover for a long time; agents could not reach either
|