@kolbo/mcp 1.76.5 → 1.77.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kolbo/mcp",
3
- "version": "1.76.5",
3
+ "version": "1.77.0",
4
4
  "description": "Kolbo AI MCP Server - Generate images, videos, music, speech, and sound effects from Claude Code",
5
5
  "main": "src/index.js",
6
6
  "bin": {
@@ -110,7 +110,6 @@ These elevate rich cinematic / reference-anchored sequences. For a short, tight,
110
110
  ## Dialogue & expression
111
111
 
112
112
  - Dialogue goes in quotes and may be in ANY language (Hebrew included). For silent tension, deliver it as expression, not speech: `He does not speak. His expression clearly says: "…"`.
113
- - **Seedance PERFORMS quoted dialogue natively** — synced voices, lip movement, and room tone come out of the video model itself. Never route scene dialogue through TTS (`generate_speech`) or `generate_lipsync`; write each line in quotes inside its shot beat (`DANIEL says: "…"`) and generate once.
114
113
 
115
114
  ## Content tone
116
115
 
@@ -9,7 +9,7 @@ Load this file when the user wants a **Seedance 2.5** video (they said "2.5" / "
9
9
 
10
10
  **Kolbo MCP routing:** `generate_video` or `generate_elements` (refs / Visual DNA / first-last). Run `list_models({ type: "text_to_video" })` and pick the Seedance 2.5 variant by name.
11
11
 
12
- **Audio:** Seedance 2.5 still emits real synced audio. `list_models` may show `sound_generation_type: none` because there is no in-app toggle (`sound_baked_in: true`). Do not tell the user the model is silent. Quoted dialogue in the prompt is PERFORMED — synced voices, lip movement, room tone — so scene dialogue never goes through `generate_speech` or `generate_lipsync`; write the lines in quotes inside their shot beats.
12
+ **Audio:** Seedance 2.5 still emits real synced audio. `list_models` may show `sound_generation_type: none` because there is no in-app toggle (`sound_baked_in: true`). Do not tell the user the model is silent.
13
13
 
14
14
  ## What's NEW in 2.5 (verified — never hedge)
15
15
 
package/src/apps/theme.js CHANGED
@@ -144,31 +144,20 @@ body {
144
144
  .k-skel.portrait { aspect-ratio: 3 / 4; }
145
145
  .k-skel::after {
146
146
  content: ''; position: absolute; inset: 0;
147
- /* The shimmer covers the WHOLE cell and outlives the skeleton — .k-skel.done
148
- only clears its paint, not its box. A pseudo-element hit-tests as its
149
- originating element, so every click on a finished cell landed on the cell
150
- instead of the media inside it: image cells still worked (their handler is
151
- ON the cell), but a <video>'s native controls never saw a single event —
152
- play, scrub, volume and fullscreen were all dead. It is decoration; it must
153
- never take a click. */
154
- pointer-events: none;
155
147
  background: linear-gradient(90deg, transparent 0%, rgba(255,255,255,0.06) 40%, rgba(255,255,255,0.10) 50%, rgba(255,255,255,0.06) 60%, transparent 100%);
156
148
  background-size: 200% 100%;
157
149
  animation: k-sweep 1.6s ease-in-out infinite;
158
150
  }
159
151
  @keyframes k-sweep { 0% { background-position: 200% 0; } 100% { background-position: -200% 0; } }
160
152
  /* Batch grid: per-cell prompt caption + a cell that already finished */
161
- /* bottom: 0 puts this exactly where a <video> draws its control bar, so it has
162
- to be click-through too — the caption is a label, not a target. The full text
163
- stays reachable: the title tooltip moved onto the cell. */
164
- .k-skel-cap { position: absolute; left: 0; right: 0; bottom: 0; z-index: 2; pointer-events: none;
153
+ .k-skel-cap { position: absolute; left: 0; right: 0; bottom: 0; z-index: 2;
165
154
  padding: 12px 8px 6px; font-size: 10.5px; color: #fff;
166
155
  background: linear-gradient(transparent, rgba(0, 0, 0, 0.65));
167
156
  white-space: nowrap; overflow: hidden; text-overflow: ellipsis; }
168
157
  .k-skel.done::after { animation: none; background: none; }
169
158
  .k-cell-fill { width: 100%; height: 100%; object-fit: contain; display: block; background: #000; }
170
159
  .k-gen-badge {
171
- position: absolute; top: 10px; left: 10px; z-index: 2; pointer-events: none;
160
+ position: absolute; top: 10px; left: 10px; z-index: 2;
172
161
  display: inline-flex; align-items: center; gap: 6px;
173
162
  padding: 4px 10px; border-radius: 999px;
174
163
  background: rgba(0, 0, 0, 0.65); /* no backdrop-filter: nested blur over the
@@ -586,12 +586,8 @@ function fillBatchCell(sc, i, g) {
586
586
  cell.classList.add('done');
587
587
  var cap = (sc.prompts && sc.prompts[i])
588
588
  ? '<span class="k-skel-cap" title="' + esc(sc.prompts[i]) + '">' + esc(sc.prompts[i]) + '</span>' : '';
589
- if (sc.prompts && sc.prompts[i]) cell.title = sc.prompts[i];
590
- // The controls attribute is not optional here: this cell is a FINISHED result
591
- // the user is meant to watch, and without it the tile was a muted poster with
592
- // no way to play, seek or unmute until the whole batch finished and repainted.
593
589
  cell.innerHTML = (sc.kind === 'video'
594
- ? '<video class="k-cell-fill" src="' + esc(u) + '"' + (r.thumbnail_url ? ' poster="' + esc(r.thumbnail_url) + '"' : '') + ' controls playsinline preload="metadata"></video>'
590
+ ? '<video class="k-cell-fill" src="' + esc(u) + '"' + (r.thumbnail_url ? ' poster="' + esc(r.thumbnail_url) + '"' : '') + ' muted playsinline preload="metadata"></video>'
595
591
  : '<img class="k-cell-fill" src="' + esc(u) + '" alt="">') + cap;
596
592
  window.kolbo.notifySize();
597
593
  }
@@ -668,21 +664,15 @@ function renderAudio(sc, urls) {
668
664
  var titleBase = track.title || sc.title || (TOOL_TITLES[sc.tool] || 'Audio');
669
665
  var title = titleBase + (urls.length > 1 ? ' — Track ' + (i + 1) : '');
670
666
  var duration = track.duration != null ? track.duration : sc.duration;
671
- // The voice's own portrait is the artwork for speech a generic note glyph
672
- // told the user nothing about the one thing that defines the take.
673
- var artwork = track.thumbnail_url || sc.thumbnail_url || sc.voice_thumbnail;
674
- var placeholder = sc.tool === 'generate_speech' ? ICONS.mic : ICONS.audio;
667
+ var artwork = track.thumbnail_url || sc.thumbnail_url;
675
668
  return '<div class="k-audio-row k-generated-audio">' +
676
- (artwork ? '<img class="k-audio-art" src="' + esc(artwork) + '" alt="" loading="lazy" onerror="this.style.display=\\'none\\'">' :
677
- '<div class="k-audio-art k-audio-placeholder">' + placeholder + '</div>') +
669
+ (artwork ? '<img class="k-audio-art" src="' + esc(artwork) + '" alt="" loading="lazy">' :
670
+ '<div class="k-audio-art k-audio-placeholder">' + ICONS.audio + '</div>') +
678
671
  '<div class="k-audio-meta"><div class="k-audio-title">' + esc(title) + '</div>' +
679
- // Voice FIRST where there is one on a speech row, who is speaking is the
680
- // thing that defines the take; the engine is secondary. Resolved names
681
- // only: a per-track model field is the raw id, and every track in one
682
- // generation came from the same model anyway.
683
- '<div class="k-audio-sub">' +
684
- [voiceLabel(sc), modelLabel(sc) || track.model, duration ? fmtDur(duration) : '']
685
- .filter(Boolean).map(esc).join(' · ') + '</div></div>' +
672
+ // Resolved name first: a per-track model field is the raw id, and every
673
+ // track in one generation came from the same model anyway.
674
+ '<div class="k-audio-sub">' + esc(modelLabel(sc) || track.model || '') +
675
+ (duration ? ' · ' + fmtDur(duration) : '') + '</div></div>' +
686
676
  '<button class="k-btn k-audio-download" data-audio-download="' + esc(u) +
687
677
  '" aria-label="Download ' + esc(title) + '">' + ICONS.download + ' Download</button>' +
688
678
  '<audio class="k-audio-player" src="' + esc(u) + '" controls preload="none" aria-label="Play ' +
@@ -777,8 +767,7 @@ function renderBatchGrid(sc) {
777
767
  var shape = items[0].type === 'video' ? 'video' : 'square';
778
768
  el('stage').innerHTML = '<div class="k-gen-grid n' + Math.min(items.length, 4) + '">' +
779
769
  items.map(function (it, i) {
780
- return '<div class="k-skel done ' + shape + '" data-focus="' + i + '"' +
781
- (it.label ? ' title="' + esc(it.label) + '"' : '') + '>' +
770
+ return '<div class="k-skel done ' + shape + '" data-focus="' + i + '">' +
782
771
  (it.type === 'video'
783
772
  ? '<video class="k-cell-fill" src="' + esc(it.url) + '" controls playsinline preload="metadata"></video>'
784
773
  : '<img class="k-cell-fill" src="' + esc(it.url) + '" alt="" loading="lazy" style="cursor:zoom-in">') +
@@ -814,8 +803,7 @@ function renderStatusGrid(sc) {
814
803
  // file — the tool only knows a url came back, not what kind it is.
815
804
  it.kind = refKind(it.url, 'image');
816
805
  var shape = it.kind === 'video' ? 'video' : 'square';
817
- return '<div class="k-skel done ' + shape + '" data-focus="' + i + '"' +
818
- (it.title ? ' title="' + esc(it.title) + '"' : '') + '>' +
806
+ return '<div class="k-skel done ' + shape + '" data-focus="' + i + '">' +
819
807
  (it.kind === 'video'
820
808
  ? '<video class="k-cell-fill" src="' + esc(it.url) + '" controls playsinline preload="metadata"></video>'
821
809
  : it.kind === 'audio'
@@ -611,7 +611,6 @@ async function uiGenerating(p) {
611
611
  ? { failed_submissions: p.failed_submissions } : {}),
612
612
  ...(p.warning ? { _warning: p.warning } : {}),
613
613
  _widget_note: 'A live Kolbo widget is rendering this generation for the user (progress + final result + action buttons). Tell the user it is generating and the card above will update — do NOT poll in a loop. If you need the output URLs (e.g. for a follow-up edit or a report), call get_generation_status ONCE with wait=true — it blocks until done. Tracking several generations? Pass ALL their ids in generation_ids in that one call.',
614
- _paid_note: 'This generation is RUNNING and the user is paying for it. If you now realize the tool, model, or parameters were wrong, call cancel_generation with this generation_id FIRST, then start the replacement — never leave a wrong generation running alongside its retry (the user gets two cards and two charges).',
615
614
  }, null, 2);
616
615
  return uiResult(UI.generation, text, structured);
617
616
  }
@@ -660,14 +659,6 @@ async function uiCompleted(p, textPayload, extraContent) {
660
659
  // above which assume everything finished together. Only set when the
661
660
  // caller actually has this shape; every existing caller is unaffected.
662
661
  ...(Array.isArray(p.items) ? { items: p.items } : {}),
663
- // The voice, by name and portrait. uiGenerating has carried this since the
664
- // chips were introduced; uiCompleted never did, so it silently dropped a
665
- // resolved voice its caller had already looked up — every FINISHED speech
666
- // card fell back to `settings.voice`, printing a raw ElevenLabs id where the
667
- // name belongs and rendering the generic note placeholder instead of the
668
- // voice's face. Exactly the "never show a raw id on a card" rule, broken on
669
- // the one path the user actually ends up looking at.
670
- ...(p.voice ? { voice_name: p.voice.name, voice_thumbnail: p.voice.thumbnail } : {}),
671
662
  // The RAW generation state, when the caller has one. `phase` above is
672
663
  // hardcoded 'completed' — it means "this tool call finished", not "the
673
664
  // generation finished" — so a status check on a still-running job looked
@@ -541,7 +541,7 @@ function registerGenerateTools(server, client, options = {}) {
541
541
  // retired textToVideoGeneration path and was stale.
542
542
  server.tool(
543
543
  'generate_video',
544
- 'Generate a video from a text prompt using Kolbo AI. For SEVERAL different videos, pass all their prompts in `prompts` in ONE call (one combined widget) — never a series of separate calls. For animating an existing still image into motion, use generate_video_from_image instead. For a coordinated multi-scene video campaign, use generate_creative_director with workflow_type="video". Supports reference images (for style/composition guidance) and Visual DNA for character consistency. ROUTE BEFORE CALLING: when reference images anchor IDENTITY (specific characters, a specific product, a location that must match) — especially 2+ of them — that is generate_elements, not this tool; reference_images here are loose style/composition hints. Decide the right tool FIRST: a mis-routed call still starts a PAID generation, and switching tools afterwards without cancel_generation leaves the user paying for both. Returns the final video URL when complete.',
544
+ 'Generate a video from a text prompt using Kolbo AI. For SEVERAL different videos, pass all their prompts in `prompts` in ONE call (one combined widget) — never a series of separate calls. For animating an existing still image into motion, use generate_video_from_image instead. For a coordinated multi-scene video campaign, use generate_creative_director with workflow_type="video". Supports reference images (for style/composition guidance) and Visual DNA for character consistency. Returns the final video URL when complete.',
545
545
  {
546
546
  prompt: z.string().optional().describe('Text description of the video to generate. Required unless `prompts` is provided.'),
547
547
  prompts: promptsField('videos'),
@@ -855,19 +855,11 @@ function registerGenerateTools(server, client, options = {}) {
855
855
  if (poll.timedOut) return poll.timedOut;
856
856
  const result = poll.result;
857
857
 
858
- // What ACTUALLY ran, not what was asked for. The status result reports the
859
- // engine's own choice (`model: "google_tts"`, `voice: "he-IL-Chirp3-HD-Kore"`)
860
- // and for an omitted model that is the ONLY place the real answer appears —
861
- // the card was labelling those "Smart Select", which is not even a text-to-
862
- // speech option, and naming the voice the caller typed rather than the one
863
- // that spoke. Same resolution addDisplayNames does for the polling path.
864
- const ranVoice = (await voiceInfo(client, result.result?.voice).catch(() => null)) || voiceRecord;
865
858
  return uiCompleted({
866
- tool: 'generate_speech', kind: 'audio', gen, client, prompt: text,
867
- model: result.result?.model || model,
868
- voice: ranVoice,
859
+ tool: 'generate_speech', kind: 'audio', gen, client, model, prompt: text,
860
+ voice: voiceRecord,
869
861
  settings: {
870
- voice: (ranVoice && ranVoice.name) || voice || 'Rachel',
862
+ voice: voice || 'Rachel',
871
863
  style: selected_style || emotion || style_instructions_preset_id || style_instructions,
872
864
  speaking_speed,
873
865
  language,
@@ -1189,10 +1181,10 @@ function registerGenerateTools(server, client, options = {}) {
1189
1181
  {
1190
1182
  prompt: z.string().describe('Locked Intro prompt (Seedance/Elements): Total line, [GLOBAL LOOK], [CAST] with @ExactDNAName for every visual_dna_ids entry, [LOCATION], then SHOT N. Not SCENE CONTEXT/OPTICS/ACTION packs. Never substitute "the left man" or "Zohar\'s" for @Name.'),
1191
1183
  model: z.string().optional().describe('Model identifier. If the user already named a family (Grok / Kling / Veo / Seedance / …), pass THAT family — never default to Seedance because Elements often uses it. Use list_models type="elements" for exact ids and elements_max_* caps. Do NOT omit (omitting = Smart Select).'),
1192
- reference_images: z.array(z.string()).optional().describe('Array of image references (product shots, character references, etc.). Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_images` URLs from list_models for the chosen model — exceeding it is a deterministic 400.**'),
1193
- reference_videos: z.array(z.string()).optional().describe('Array of reference videos for models that accept video inputs. Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_videos` URLs from list_models — if the cap is 0 the model rejects videos.**'),
1194
- reference_audio_urls: z.array(z.string()).optional().describe('Array of reference audio tracks for models that accept audio inputs. Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_audio` URLs from list_models.** `audio_url` remains supported as the legacy single-track form.'),
1195
- audio_url: z.string().optional().describe('A single reference audio track — legacy form of reference_audio_urls. Accepts a public URL (forwarded as-is; if the API rejects an external URL as untrusted, it is auto-rehosted into the media library and retried once) OR an absolute local path, which is uploaded for you. **Audio constraints: `elements_max_audio` from list_models gates whether audio is accepted at all; audio duration must fall within `min_audio_duration`-`max_audio_duration`; format must be in `supported_audio_formats` (if specified).**'),
1184
+ reference_images: z.array(z.string()).optional().describe('Array of image references (product shots, character references, etc.). Accepts a public URL (forwarded as-is, never re-uploaded) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_images` URLs from list_models for the chosen model — exceeding it is a deterministic 400.**'),
1185
+ reference_videos: z.array(z.string()).optional().describe('Array of reference videos for models that accept video inputs. Accepts a public URL (forwarded as-is, never re-uploaded) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_videos` URLs from list_models — if the cap is 0 the model rejects videos.**'),
1186
+ reference_audio_urls: z.array(z.string()).optional().describe('Array of reference audio tracks for models that accept audio inputs. Accepts a public URL (forwarded as-is, never re-uploaded) OR an absolute local path, which is uploaded for you. **Cap: pass at most `elements_max_audio` URLs from list_models.** `audio_url` remains supported as the legacy single-track form.'),
1187
+ audio_url: z.string().optional().describe('A single reference audio track — legacy form of reference_audio_urls. Accepts a public URL (forwarded as-is, never re-uploaded) OR an absolute local path, which is uploaded for you. **Audio constraints: `elements_max_audio` from list_models gates whether audio is accepted at all; audio duration must fall within `min_audio_duration`-`max_audio_duration`; format must be in `supported_audio_formats` (if specified).**'),
1196
1188
  files: z.array(z.string()).optional().describe('Untyped catch-all for mixed media — images, videos AND audio, each a URL or an absolute local path. The kind is detected from the file extension and the item is routed to the matching reference list, so a local .mp4 is sent as a video and a local .mp3 as audio. Prefer the typed lists (reference_images / reference_videos / reference_audio_urls) when you already know the kind; they accept local paths too. URLs given here are forwarded as URLs, never re-uploaded. **Caps still apply per kind: `elements_max_images` / `elements_max_videos` / `elements_max_audio` from list_models. Local uploads are capped at 200MB each.**'),
1197
1189
  duration: z.number().optional().describe('Output duration in seconds. Must be in `supported_durations` from list_models, OR within `min_output_duration`-`max_output_duration`. Default: 5'),
1198
1190
  aspect_ratio: z.string().optional().describe('Aspect ratio (e.g., "16:9", "9:16", "1:1"). Must be in `supported_aspect_ratios` from list_models. Default: "16:9"'),
@@ -1258,10 +1250,10 @@ function registerGenerateTools(server, client, options = {}) {
1258
1250
  session_name, project_id, session_id
1259
1251
  };
1260
1252
 
1261
- const startElements = async () => {
1262
- if (!locals.length) {
1263
- return client.post('/v1/generate/elements', body);
1264
- }
1253
+ let startResponse;
1254
+ if (!locals.length) {
1255
+ startResponse = await client.post('/v1/generate/elements', body);
1256
+ } else {
1265
1257
  const resolved = await Promise.all(locals.map(({ src, kind }) =>
1266
1258
  resolveToBuffer(src, kind, { maxBytes: ELEMENTS_MAX_UPLOAD_BYTES })));
1267
1259
  const form = new FormData();
@@ -1285,36 +1277,7 @@ function registerGenerateTools(server, client, options = {}) {
1285
1277
  for (const f of resolved) {
1286
1278
  form.append('files', f.buffer, { filename: f.filename, contentType: f.contentType });
1287
1279
  }
1288
- return client.postMultipart('/v1/generate/elements', form);
1289
- };
1290
-
1291
- let startResponse;
1292
- try {
1293
- startResponse = await startElements();
1294
- } catch (err) {
1295
- // The elements trust gate only accepts remote references that are BOTH
1296
- // Kolbo-hosted AND registered to this caller's library/project — any
1297
- // other public URL 400s with this code, even though every schema here
1298
- // says "public URL, forwarded as-is". Agents used to recover by hand
1299
- // (upload_media, then resend). Do that detour for them: rehost every
1300
- // remote reference into the caller's library and retry ONCE. The gate
1301
- // fires before any charge, so the failed first attempt costs nothing.
1302
- if (err?.code !== 'UNTRUSTED_REFERENCE_MEDIA_URL') throw err;
1303
- for (const kind of ['image', 'video', 'audio']) {
1304
- media[kind] = await Promise.all(media[kind].map(async (src) => {
1305
- if (!isUrlSource(src)) return src;
1306
- const file = await resolveToBuffer(src, kind, { maxBytes: ELEMENTS_MAX_UPLOAD_BYTES });
1307
- const form = new FormData();
1308
- form.append('file', file.buffer, { filename: file.filename, contentType: file.contentType });
1309
- if (project_id) form.append('project_id', project_id);
1310
- const uploaded = await client.postMultipart('/v1/media/upload', form);
1311
- return uploaded?.media?.url || uploaded?.url || src;
1312
- }));
1313
- }
1314
- body.reference_images = some(urlsOf('image'));
1315
- body.reference_videos = some(urlsOf('video'));
1316
- body.reference_audio_urls = some(urlsOf('audio'));
1317
- startResponse = await startElements();
1280
+ startResponse = await client.postMultipart('/v1/generate/elements', form);
1318
1281
  }
1319
1282
 
1320
1283
  if (ui()) return uiGenerating({
@@ -1613,18 +1576,10 @@ function registerGenerateTools(server, client, options = {}) {
1613
1576
  highlighted: z.object({ font: z.string().optional(), weight: z.number().int().min(100).max(900).optional(), color: z.string().optional() }).optional().describe('Highlighted word tier styling.'),
1614
1577
  }).optional(),
1615
1578
  }).optional().describe('VEED Subtitles only: style overrides. Any omitted field keeps the preset default. Best supported by Basic presets.'),
1616
- enhancement_model: z.string().optional()
1617
- .describe('Topaz Slow Motion only (model "topaz/interpolate/video"): retiming engine — "Apollo" (smooth motion, default), "Chronos" (complex motion and occlusion), or "Aion" (highest quality for extreme slow motion, and the most expensive).'),
1618
- target_fps: z.number().optional()
1619
- .describe('Topaz Slow Motion only: frames per second of the OUTPUT (16-120, default 60). Higher rates generate more frames and cost proportionally more.'),
1620
- slowdown_factor: z.number().optional()
1621
- .describe('Topaz Slow Motion only: how many times longer the output runs (1-8, default 1). 4 turns a 5s clip into 20s of slow motion. Billing is on the OUTPUT length, so an 8x pass costs 8x a 1x pass.'),
1622
- output_format: z.string().optional()
1623
- .describe('Topaz HDR only (model "topaz/sdr-to-hdr/video"): "mp4" for 10-bit H.265 HDR10 (default) or "prores" for 10-bit ProRes 422 HQ. Both are HDR masters — the in-app player shows a tone-mapped SDR preview and the HDR file is the download.'),
1624
1579
  project_id: projectIdField,
1625
1580
  session_id: sessionIdField
1626
1581
  },
1627
- async ({ source_video, prompt, model, aspect_ratio, duration, enhance_prompt = false, visual_dna_ids, resolution, sound_enabled, reference_images, reference_videos, elements, preset, source_language, translation_language, srt_content, srt_file_url, vocabulary, customization, enhancement_model, target_fps, slowdown_factor, output_format, project_id, session_id }) => {
1582
+ async ({ source_video, prompt, model, aspect_ratio, duration, enhance_prompt = false, visual_dna_ids, resolution, sound_enabled, reference_images, reference_videos, elements, preset, source_language, translation_language, srt_content, srt_file_url, vocabulary, customization, project_id, session_id }) => {
1628
1583
  model = await canonicalModelId(client, model, 'video_to_video'); // lenient id resolution ("z-image" → "z-image/turbo")
1629
1584
  if (!source_video) throw new Error('source_video is required');
1630
1585
 
@@ -1634,9 +1589,7 @@ function registerGenerateTools(server, client, options = {}) {
1634
1589
  startResponse = await client.post('/v1/generate/video-from-video', {
1635
1590
  video_url: source_video, prompt, model, aspect_ratio, duration, enhance_prompt, visual_dna_ids, resolution, sound_enabled,
1636
1591
  reference_images, reference_videos, elements, preset, source_language, translation_language,
1637
- srt_content, srt_file_url, vocabulary, customization,
1638
- enhancement_model, target_fps, slowdown_factor, output_format,
1639
- project_id, session_id
1592
+ srt_content, srt_file_url, vocabulary, customization, project_id, session_id
1640
1593
  });
1641
1594
  } else {
1642
1595
  const resolved = await resolveToBuffer(source_video, 'video');
@@ -1660,10 +1613,6 @@ function registerGenerateTools(server, client, options = {}) {
1660
1613
  if (reference_images) form.append('reference_images', JSON.stringify(reference_images));
1661
1614
  if (reference_videos) form.append('reference_videos', JSON.stringify(reference_videos));
1662
1615
  if (elements) form.append('elements', JSON.stringify(elements));
1663
- if (enhancement_model) form.append('enhancement_model', enhancement_model);
1664
- if (target_fps !== undefined) form.append('target_fps', String(target_fps));
1665
- if (slowdown_factor !== undefined) form.append('slowdown_factor', String(slowdown_factor));
1666
- if (output_format) form.append('output_format', output_format);
1667
1616
  if (project_id) form.append('project_id', project_id);
1668
1617
  if (session_id) form.append('session_id', session_id);
1669
1618
  startResponse = await client.postMultipart('/v1/generate/video-from-video', form);
@@ -1873,8 +1822,7 @@ function registerGenerateTools(server, client, options = {}) {
1873
1822
  'camera_angle',
1874
1823
  'split', 'split_upscale',
1875
1824
  'multi_shot',
1876
- 'magic_edit',
1877
- 'enhance'
1825
+ 'magic_edit'
1878
1826
  ]).describe([
1879
1827
  'Edit operation:',
1880
1828
  '"upscale" — increase resolution by 2×, 3×, or 4× (use `scale`). "clarity_upscale" — AI-powered clarity upscale with detail enhancement (use `resolution`).',
@@ -1883,7 +1831,6 @@ function registerGenerateTools(server, client, options = {}) {
1883
1831
  '"removebg" — remove the image background, output is transparent PNG.',
1884
1832
  '"background_replace" — remove background and replace it with AI-generated content from `prompt`.',
1885
1833
  '"enhance_skin" — portrait skin retouching (use `skin_strength`: "subtle" | "realistic" | "pimple" | "freckle").',
1886
- '"enhance" — Topaz photo correction at the SOURCE resolution (no resizing). Pick the tool with `model`: "topaz/adjust/image" (exposure, white balance, or colorizing a black-and-white photo), "topaz/sharpen/image" (lens / motion / portrait / wildlife blur, or Super Focus for severely blurred shots), "topaz/denoise/image" (high-ISO and night noise), "topaz/restore/image" (old or damaged photos, dust and scratches). Choose the specific engine with `enhancement_model` — call list_models type="graphics_enhance" to see each model\'s engines. To make an image BIGGER use "upscale" instead.',
1887
1834
  '"inpaint" — paint over a masked area using `mask_image_url` (B&W mask, white = fill area) and optional `prompt`. Add reference images via `additional_images`.',
1888
1835
  '"erase" — erase an object defined by `mask_image_url` (white = erase area).',
1889
1836
  '"face_swap" — swap the face in `image_url` with the face from `mask_image_url` (required).',
@@ -1898,13 +1845,7 @@ function registerGenerateTools(server, client, options = {}) {
1898
1845
 
1899
1846
  // ── upscale ────────────────────────────────────────────
1900
1847
  scale: z.number().optional()
1901
- .describe('Upscale factor: 1, 2, 4 or 8. Used with operation="upscale". Default: 2.'),
1902
-
1903
- enhancement_model: z.string().optional()
1904
- .describe('Topaz engine. With operation="upscale" on model "topaz/upscale/image" it selects the engine family: "Standard V2" / "High Fidelity V3" / "CGI" / "Text Refine" (faithful), "Wonder 3.5" (rebuilds natural detail), "Bloom 2" (reinvents detail — most expensive), "Transparent" (keeps the alpha channel). With operation="enhance" it selects the correction engine for the chosen model. Omit for the model default. Call list_models to see the engines a model offers.'),
1905
-
1906
- output_format: z.string().optional()
1907
- .describe('Output image format: "jpeg" or "png". Used with Topaz "upscale" and "enhance". Defaults to jpeg (the transparent upscaler always returns png).'),
1848
+ .describe('Upscale factor: 2, 3, or 4. Used with operation="upscale". Default: 2.'),
1908
1849
 
1909
1850
  resolution: z.string().optional()
1910
1851
  .describe('Target output resolution (e.g. "4k", "2k", "1080p"). Used with "clarity_upscale", "split_upscale", "multi_shot".'),
@@ -1961,7 +1902,6 @@ function registerGenerateTools(server, client, options = {}) {
1961
1902
  image_url, operation, model, scale, aspect_ratio, skin_strength, prompt,
1962
1903
  mask_image_url, additional_images, generate_all_angles, resolution, quality, ai_optimize = false,
1963
1904
  zoom_out_percentage, expand_left, expand_right, expand_top, expand_bottom,
1964
- enhancement_model, output_format,
1965
1905
  project_id, session_id
1966
1906
  }) => {
1967
1907
  // No `type` argument: these are operation-routed tools (upscale / reframe /
@@ -1971,9 +1911,6 @@ function registerGenerateTools(server, client, options = {}) {
1971
1911
 
1972
1912
  // Basic validation
1973
1913
  if (operation === 'reframe' && !aspect_ratio) throw new Error('aspect_ratio is required for reframe');
1974
- if (operation === 'enhance' && !model) {
1975
- throw new Error('model is required for enhance — pick one of: topaz/adjust/image, topaz/sharpen/image, topaz/denoise/image, topaz/restore/image');
1976
- }
1977
1914
  if (operation === 'background_replace' && !prompt) throw new Error('prompt is required for background_replace');
1978
1915
  if (operation === 'face_swap' && !mask_image_url && !(additional_images && additional_images.length > 0)) {
1979
1916
  throw new Error('mask_image_url (face reference) is required for face_swap');
@@ -1983,7 +1920,6 @@ function registerGenerateTools(server, client, options = {}) {
1983
1920
  image_url, operation, model, scale, aspect_ratio, skin_strength, prompt,
1984
1921
  mask_image_url, additional_images, generate_all_angles, resolution, quality, ai_optimize,
1985
1922
  zoom_out_percentage, expand_left, expand_right, expand_top, expand_bottom,
1986
- enhancement_model, output_format,
1987
1923
  project_id, session_id
1988
1924
  });
1989
1925
 
@@ -2056,8 +1992,6 @@ function registerGenerateTools(server, client, options = {}) {
2056
1992
  .describe('Target resolution (e.g. "4k", "2k", "1080p"). Used with "upscale" and "reframe".'),
2057
1993
  target_fps: z.number().optional()
2058
1994
  .describe('Target frame rate (e.g. 24, 30, 60). Used with operation="upscale".'),
2059
- enhancement_model: z.string().optional()
2060
- .describe('Topaz enhancement engine for operation="upscale" on model "topaz/upscale/video". Families: Precision (faithful — "Proteus", "Artemis High Quality", "Iris", "Dione TV", "Gaia CG", "Gaia 2"), Denoise ("Nyx", "Nyx Fast"), Generative (rebuilds detail that is not in the source — "Starlight Fast 2", "Starlight HQ", "Starlight Precise 2.6"), Creative ("Astra 2", always renders 4K). Generative and Creative engines cost several times the Precision rate. Omit for the default.'),
2061
1995
 
2062
1996
  // ── reframe ────────────────────────────────────────────
2063
1997
  aspect_ratio: z.string().optional()
@@ -2121,7 +2055,7 @@ function registerGenerateTools(server, client, options = {}) {
2121
2055
  async ({
2122
2056
  video_url, operation, model, aspect_ratio, scale, prompt,
2123
2057
  image_url, audio_url, duration, mode,
2124
- target_fps, resolution, enhancement_model,
2058
+ target_fps, resolution,
2125
2059
  grid_position_x, grid_position_y,
2126
2060
  sound_effect_prompt, background_music_prompt, original_sound, cfg_strength,
2127
2061
  refine_edges, subject_is_person,
@@ -2144,7 +2078,7 @@ function registerGenerateTools(server, client, options = {}) {
2144
2078
  if (operation === 'extend' && !duration) throw new Error('duration is required for extend');
2145
2079
 
2146
2080
  const gen = await client.post('/v1/edit/video', {
2147
- video_url, operation, model, aspect_ratio, scale, prompt, enhancement_model,
2081
+ video_url, operation, model, aspect_ratio, scale, prompt,
2148
2082
  image_url, audio_url, duration, mode,
2149
2083
  target_fps, resolution,
2150
2084
  grid_position_x, grid_position_y,