@kolbo/mcp 1.57.0 → 1.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kolbo/mcp",
3
- "version": "1.57.0",
3
+ "version": "1.58.0",
4
4
  "description": "Kolbo AI MCP Server - Generate images, videos, music, speech, and sound effects from Claude Code",
5
5
  "main": "src/index.js",
6
6
  "bin": {
@@ -10,8 +10,9 @@
10
10
  "start": "node src/index.js",
11
11
  "smoke": "node scripts/smoke.js",
12
12
  "check-parity": "node scripts/check-parity.js",
13
- "prepublishOnly": "node scripts/smoke.js && node scripts/check-parity.js && node scripts/check-widget-fields.js && node scripts/check-skill-tools.js && node scripts/check-install.js",
13
+ "prepublishOnly": "node scripts/smoke.js && node scripts/check-parity.js && node scripts/check-widget-fields.js && node scripts/check-widget-render.js && node scripts/check-skill-tools.js && node scripts/check-install.js",
14
14
  "check-widget-fields": "node scripts/check-widget-fields.js",
15
+ "check-widget-render": "node scripts/check-widget-render.js",
15
16
  "check-skill-tools": "node scripts/check-skill-tools.js",
16
17
  "check-install": "node scripts/check-install.js"
17
18
  },
package/src/apps/theme.js CHANGED
@@ -180,7 +180,7 @@ body {
180
180
  display: inline-flex; align-items: center; justify-content: center;
181
181
  cursor: pointer; opacity: 0; transition: opacity 150ms var(--smooth), background 150ms var(--smooth);
182
182
  }
183
- .k-media:hover .k-dl, .k-viewer:hover .k-dl { opacity: 1; }
183
+ .k-media:hover .k-dl, .k-viewer:hover .k-dl, .k-skel:hover .k-dl { opacity: 1; }
184
184
  .k-dl:hover { background: var(--brand); border-color: var(--brand); }
185
185
  .k-viewer { position: relative; }
186
186
 
@@ -14,7 +14,8 @@ const { widgetPage } = require('../html');
14
14
  * status_args, // extra args for the poll tool (optional)
15
15
 
16
16
  * model, model_icon, prompt, count,
17
- * settings: { duration, resolution, aspect_ratio, audio, voice, mode },
17
+ * settings: { duration, resolution, aspect_ratio, quality, audio, voice, mode,
18
+ * enhance_prompt, web_search, visual_dna, moodboard, preset, cinematic },
18
19
  * reference_image, // thumbnail URL (optional)
19
20
  * urls, thumbnail_url, title, duration, credits_used,
20
21
  * tracks: [{ title, duration, thumbnail_url, model }], // optional audio metadata by URL index
@@ -111,11 +112,18 @@ function renderChips(sc) {
111
112
  if (s.duration) h += chip(ICONS.clock + ' ' + fmtDur(s.duration));
112
113
  if (s.resolution) h += chip(esc(s.resolution));
113
114
  if (s.aspect_ratio) h += chip(esc(s.aspect_ratio));
115
+ if (s.quality) h += chip(esc(s.quality) + ' quality');
116
+ if (s.enhance_prompt) h += chip(ICONS.sparkle + ' enhanced');
117
+ if (s.web_search) h += chip('web search');
118
+ if (s.visual_dna) h += chip(s.visual_dna + ' Visual DNA');
119
+ if (s.moodboard) h += chip('moodboard');
120
+ if (s.preset) h += chip('preset');
121
+ if (s.cinematic) h += chip('cinematic');
114
122
  if (s.audio) h += chip(ICONS.sound + ' audio');
115
123
  if (s.voice) h += chip(ICONS.mic + ' ' + esc(s.voice));
116
124
  if (s.mode) h += chip(esc(s.mode));
117
125
  if (sc.count > 1) h += chip('×' + sc.count);
118
- if (sc.reference_image) h += '<img class="k-ref-thumb" src="' + esc(sc.reference_image) + '" alt="" title="Reference image" onerror="this.style.display=\\'none\\'">';
126
+ if (sc.reference_image) h += '<img class="k-ref-thumb" src="' + esc(sc.reference_image) + '" alt="" loading="lazy" title="Reference image" onerror="this.style.display=\\'none\\'">';
119
127
  el('chips').innerHTML = h;
120
128
  }
121
129
  function chip(inner) { return '<span class="k-chip">' + inner + '</span>'; }
@@ -237,13 +245,48 @@ var MAX_POLL_ERRORS = 30;
237
245
  var pollStart = 0, pollErrors = 0;
238
246
  var cancelRequested = false; // set by the Stop button; freezes the poll loop
239
247
 
248
+ /* ---------- offscreen gate ----------
249
+ The host mounts one of these iframes per generation, and re-delivers the
250
+ ORIGINAL "submitted" (phase:generating) result on every conversation open —
251
+ so a 50-generation session used to fire 50 status tools/call round trips plus
252
+ 50+ full-resolution media downloads before the user had scrolled to any of
253
+ them. Hold the FIRST poll (and therefore every media request the result
254
+ produces) until the card is actually on screen. Once a card has been seen it
255
+ polls normally forever — a live generation the user scrolls away from still
256
+ finishes and still reports back. */
257
+ var seen = false, whenSeenFns = [];
258
+ function releaseSeen() {
259
+ if (seen) return;
260
+ seen = true;
261
+ var fns = whenSeenFns; whenSeenFns = [];
262
+ fns.forEach(function (f) { try { f(); } catch (e) {} });
263
+ }
264
+ (function () {
265
+ var card = document.querySelector('.k-card');
266
+ if (!window.IntersectionObserver || !card) return releaseSeen();
267
+ var fired = false;
268
+ var io = new IntersectionObserver(function (entries) {
269
+ fired = true;
270
+ if (!entries.some(function (e) { return e.isIntersecting; })) return;
271
+ io.disconnect();
272
+ releaseSeen();
273
+ // IO clips against ancestor frames, so this is true parent-viewport
274
+ // visibility. rootMargin starts the work just before the card scrolls in.
275
+ }, { rootMargin: '400px' });
276
+ io.observe(card);
277
+ // A host where IO never reports at all must not strand the card forever.
278
+ setTimeout(function () { if (!fired) releaseSeen(); }, 8000);
279
+ })();
280
+
240
281
  function schedulePoll(sc) {
241
282
  if (cancelRequested) return;
283
+ if (!seen) { whenSeenFns.push(function () { schedulePoll(sc); }); return; }
284
+ // The call itself long-waits server-side (normally up to three minutes).
285
+ // This short pause only separates successive wait windows — the FIRST call
286
+ // goes out immediately, so a card revealed by scrolling resolves at once.
287
+ var delay = pollStart ? 1500 : 0;
242
288
  if (!pollStart) pollStart = Date.now();
243
289
  clearTimeout(pollTimer);
244
- // The call itself long-waits server-side (normally up to three minutes).
245
- // This short pause only separates successive wait windows.
246
- var delay = 1500;
247
290
  pollTimer = setTimeout(function () { poll(sc); }, delay);
248
291
  }
249
292
  function poll(sc) {
@@ -304,7 +347,10 @@ function poll(sc) {
304
347
  /* ---------- batch (prompts[] fan-out) ---------- */
305
348
  // Each poll round resolves when every id has completed or its wait window
306
349
  // closed (~3 min), so finished cells fill in per round while the rest keep
307
- // their skeleton. When all_done, the set renders through the scenes viewer.
350
+ // their skeleton. When all_done the SAME grid is re-rendered from the resolved
351
+ // set — a batch is one grouped card end to end. It must NOT fall through to the
352
+ // scenes carousel: that collapses eight tiles into one big image plus a thumb
353
+ // strip, which is where the grouping (and the per-tile prompt caption) was lost.
308
354
  function handleBatchStatus(sc, st) {
309
355
  pollErrors = 0;
310
356
  var gens = st.generations || [];
@@ -336,8 +382,9 @@ function handleBatchStatus(sc, st) {
336
382
  });
337
383
  state = done;
338
384
  el('credits').textContent = done.credits_used != null ? fmtCredits(done.credits_used) : '';
339
- if (failedCount) setPhaseChip(failedCount + ' failed', false);
340
385
  renderResult(done);
386
+ // After renderResult — it resets the chip, so setting this first erased it.
387
+ if (failedCount) setPhaseChip(failedCount + ' failed', false);
341
388
  try {
342
389
  window.kolbo.updateModelContext(
343
390
  'Batch generation completed (' + (sc.tool || '') + '): ' + scenes.length + ' of ' + gens.length + ' succeeded.' +
@@ -368,6 +415,7 @@ function renderResult(sc) {
368
415
  clearTimeout(pollTimer);
369
416
 
370
417
  setPhaseChip('', false);
418
+ if (sc.batch && sc.scenes && sc.scenes.length) return renderBatchGrid(sc);
371
419
  if (sc.kind === 'scenes' && sc.scenes && sc.scenes.length) return renderScenes(sc);
372
420
  var urls = sc.urls || [];
373
421
  if (!urls.length) return renderError('No output received');
@@ -396,7 +444,7 @@ function renderImages(sc, urls) {
396
444
  var thumbs = '';
397
445
  if (urls.length > 1) {
398
446
  thumbs = '<div class="k-thumbs">' + urls.map(function (u, i) {
399
- return '<div class="k-thumb' + (i === selected ? ' active' : '') + '" data-i="' + i + '"><img src="' + esc(u) + '" alt=""></div>';
447
+ return '<div class="k-thumb' + (i === selected ? ' active' : '') + '" data-i="' + i + '"><img src="' + esc(u) + '" alt="" loading="lazy"></div>';
400
448
  }).join('') + '</div>';
401
449
  }
402
450
  el('stage').innerHTML = viewer + thumbs;
@@ -415,8 +463,10 @@ function renderImages(sc, urls) {
415
463
  }
416
464
 
417
465
  function renderVideo(sc, urls) {
466
+ // preload="none" behind a poster: the card shows the still until the user
467
+ // hits play, instead of pulling the video header on mount.
418
468
  el('stage').innerHTML = '<div class="k-viewer"><video id="main-video" src="' + esc(urls[0]) + '"' +
419
- (sc.thumbnail_url ? ' poster="' + esc(sc.thumbnail_url) + '"' : '') + ' controls playsinline></video>' +
469
+ (sc.thumbnail_url ? ' poster="' + esc(sc.thumbnail_url) + '" preload="none"' : ' preload="metadata"') + ' controls playsinline></video>' +
420
470
  dlBtnHTML(urls[0]) + '</div>';
421
471
  wireDlButtons(el('stage'));
422
472
  }
@@ -429,14 +479,14 @@ function renderAudio(sc, urls) {
429
479
  var duration = track.duration != null ? track.duration : sc.duration;
430
480
  var artwork = track.thumbnail_url || sc.thumbnail_url;
431
481
  return '<div class="k-audio-row k-generated-audio">' +
432
- (artwork ? '<img class="k-audio-art" src="' + esc(artwork) + '" alt="">' :
482
+ (artwork ? '<img class="k-audio-art" src="' + esc(artwork) + '" alt="" loading="lazy">' :
433
483
  '<div class="k-audio-art k-audio-placeholder">' + ICONS.audio + '</div>') +
434
484
  '<div class="k-audio-meta"><div class="k-audio-title">' + esc(title) + '</div>' +
435
485
  '<div class="k-audio-sub">' + esc(track.model || sc.model || '') +
436
486
  (duration ? ' · ' + fmtDur(duration) : '') + '</div></div>' +
437
487
  '<button class="k-btn k-audio-download" data-audio-download="' + esc(u) +
438
488
  '" aria-label="Download ' + esc(title) + '">' + ICONS.download + ' Download</button>' +
439
- '<audio class="k-audio-player" src="' + esc(u) + '" controls preload="metadata" aria-label="Play ' +
489
+ '<audio class="k-audio-player" src="' + esc(u) + '" controls preload="none" aria-label="Play ' +
440
490
  esc(title) + '"></audio></div>';
441
491
  }).join('');
442
492
  Array.prototype.forEach.call(el('stage').querySelectorAll('[data-audio-download]'), function (b) {
@@ -456,7 +506,7 @@ function renderAudio(sc, urls) {
456
506
 
457
507
  function render3d(sc, urls) {
458
508
  el('stage').innerHTML = (sc.thumbnail_url
459
- ? '<div class="k-viewer"><img src="' + esc(sc.thumbnail_url) + '" alt=""></div>' : '') +
509
+ ? '<div class="k-viewer"><img src="' + esc(sc.thumbnail_url) + '" alt="" loading="lazy"></div>' : '') +
460
510
  urls.map(function (u) {
461
511
  var extMatch = u.split('?')[0].match(/\\.(\\w+)$/);
462
512
  var ext = extMatch ? extMatch[1].toUpperCase() : 'FILE';
@@ -508,6 +558,32 @@ function sceneItems(sc) {
508
558
  return items;
509
559
  }
510
560
 
561
+ // Batch (prompts[] fan-out) result: the SAME tile grid the generating phase
562
+ // showed, each tile still captioned with the prompt that produced it. Downloads
563
+ // are per-tile (a batch has no single "current" url); click a tile to focus it.
564
+ function renderBatchGrid(sc) {
565
+ var items = sceneItems(sc);
566
+ if (!items.length) return renderError('No completed results received');
567
+ var shape = items[0].type === 'video' ? 'video' : 'square';
568
+ el('stage').innerHTML = '<div class="k-gen-grid n' + Math.min(items.length, 4) + '">' +
569
+ items.map(function (it, i) {
570
+ return '<div class="k-skel done ' + shape + '" data-focus="' + i + '">' +
571
+ (it.type === 'video'
572
+ ? '<video class="k-cell-fill" src="' + esc(it.url) + '" controls playsinline preload="metadata"></video>'
573
+ : '<img class="k-cell-fill" src="' + esc(it.url) + '" alt="" loading="lazy" style="cursor:zoom-in">') +
574
+ (it.label ? '<span class="k-skel-cap" title="' + esc(it.label) + '">' + esc(it.label) + '</span>' : '') +
575
+ dlBtnHTML(it.url) + '</div>';
576
+ }).join('') + '</div>';
577
+ wireDlButtons(el('stage'));
578
+ Array.prototype.forEach.call(el('stage').querySelectorAll('[data-focus]'), function (cell) {
579
+ var it = items[+cell.getAttribute('data-focus')];
580
+ if (it.type !== 'image') return; // <video controls> owns its own clicks
581
+ cell.onclick = function () { focusMedia(it.url); };
582
+ });
583
+ renderActions(sc);
584
+ window.kolbo.notifySize();
585
+ }
586
+
511
587
  function renderScenes(sc) {
512
588
  var items = sceneItems(sc);
513
589
  if (!items.length) return renderError('No completed scenes received');
@@ -555,7 +631,7 @@ function exitFocus() {
555
631
  window.kolbo.requestDisplayMode('inline').catch(function () {});
556
632
  isFullscreen = false;
557
633
  applyFullscreen(false);
558
- renderScenes(state); // restore the grid
634
+ renderResult(state); // restore whichever multi-item view we came from
559
635
  window.kolbo.notifySize();
560
636
  }
561
637
 
@@ -740,7 +816,8 @@ function completedFromPlain(sc) {
740
816
  settings: {
741
817
  duration: sc.duration || originArgs.duration,
742
818
  resolution: originArgs.resolution,
743
- aspect_ratio: originArgs.aspect_ratio
819
+ aspect_ratio: originArgs.aspect_ratio,
820
+ quality: originArgs.quality
744
821
  },
745
822
  urls: sc.urls || []
746
823
  });
@@ -73,6 +73,24 @@ async function pollBatch(client, batch, { interval, timeout }) {
73
73
  };
74
74
  }
75
75
 
76
+ // ─── Widget settings block ──────────────────────────────────────────────────
77
+ // What the CALLER actually asked for, for the generation card AND for the model
78
+ // reading the tool result. Undefined/false keys are dropped by JSON.stringify, so
79
+ // only values that were really supplied ever surface. This used to be
80
+ // `{ resolution, aspect_ratio }` only — `quality` (and every knob below it) was
81
+ // silently dropped, so three calls at low/medium/high rendered identical cards.
82
+ const imageSettings = (a = {}) => ({
83
+ resolution: a.resolution,
84
+ aspect_ratio: a.aspect_ratio,
85
+ quality: a.quality,
86
+ enhance_prompt: a.enhance_prompt || undefined,
87
+ web_search: a.enable_web_search || undefined,
88
+ visual_dna: (a.visual_dna_ids && a.visual_dna_ids.length) || undefined,
89
+ moodboard: a.moodboard_id ? true : undefined,
90
+ preset: a.preset_id ? true : undefined,
91
+ cinematic: a.cinematic ? true : undefined,
92
+ });
93
+
76
94
  const promptsField = (what) => z.array(z.string()).optional().describe(
77
95
  `BATCH MODE — several DIFFERENT prompts (2–${MAX_BATCH_PROMPTS}) generated concurrently in ONE call and rendered together in ONE combined widget. Whenever the user wants multiple distinct ${what} with their own prompts, ALWAYS pass them all here instead of making several separate calls — separate calls clutter the chat with stacked widgets. All prompts share the same model/settings. When set, \`prompt\` is ignored. For N variations of a SINGLE prompt use num_images (image tools); for an AI-planned coherent scene set use generate_creative_director.`
78
96
  );
@@ -121,7 +139,7 @@ function registerGenerateTools(server, client, options = {}) {
121
139
  const batch = await submitBatch(prompts, (p) => client.post('/v1/generate/image', { ...shared, prompt: p }));
122
140
  if (ui()) return uiGenerating({
123
141
  tool: 'generate_image', kind: 'image', gen: batch.ok[0].gen, client, model,
124
- count: batch.ids.length, settings: { resolution, aspect_ratio },
142
+ count: batch.ids.length, settings: imageSettings(shared),
125
143
  generation_ids: batch.ids, prompts: batch.ok.map((o) => o.prompt),
126
144
  failed_submissions: batch.failed,
127
145
  status_args: { generation_ids: batch.ids, wait: true },
@@ -134,7 +152,7 @@ function registerGenerateTools(server, client, options = {}) {
134
152
 
135
153
  if (ui()) return uiGenerating({
136
154
  tool: 'generate_image', kind: 'image', gen, client, model, prompt,
137
- count: num_images, settings: { resolution, aspect_ratio },
155
+ count: num_images, settings: imageSettings(shared),
138
156
  reference_image: reference_images?.[0]
139
157
  });
140
158
 
@@ -167,7 +185,7 @@ function registerGenerateTools(server, client, options = {}) {
167
185
  'THE tool for ANY prompt-driven / content edit of an existing image — changing the scene ("make it night", "change the sky to sunset"), adding/removing/replacing objects, restyling, recoloring, compositing, or any "edit this image to…" request. This is the image-editing equivalent of generate_image and runs on strong dedicated editing models (nano-banana-2, gpt-image-2). Provide the source image URL(s) in `source_images` and the instruction in `prompt`. Supports Visual DNA profiles and moodboards for style-consistent edits. Do NOT use `edit_image` for these — that tool is only for mechanical enhancements (upscale/reframe/remove-background/skin). For a brand-new image from scratch, use generate_image. Returns the edited image URL(s) when complete.',
168
186
  {
169
187
  prompt: z.string().describe('Description of the edit to apply (e.g., "remove the background", "change the sky to sunset")'),
170
- model: z.string().optional().describe('Model identifier — REQUIRED in practice: pick a specific IMAGE-EDITING model, do NOT omit (omitting = Smart Select auto-pick, which we avoid). Strong current defaults: "nano-banana-pro/edit" (best general prompt editor), "gpt-image/1.5-image-to-image" (photoreal), or "flux-2/edit". NOTE: text-to-image ids like "nano-banana-2"/"gpt-image-2" are NOT editors don\'t use them here. Call list_models type="image_editing" to see all options and pick per the user\'s intent.'),
188
+ model: z.string().optional().describe('Model identifier — REQUIRED in practice: pick a specific model, do NOT omit (omitting = Smart Select auto-pick, which we avoid). Many text-to-image ids double as editors: the server auto-routes a base id to its editing variant when source_images is present (e.g. "gpt-image-2" → gpt-image-2/edit, "nano-banana-2" → nano-banana-2/edit) — passing the bare id is fine, no need to hunt for the "/edit" suffix yourself. BUT this only works for models that actually have a registered edit variant (most flagship models do: gpt-image, nano-banana, flux-2, seedream, qwen, wan, grok-imagine, kling-image families). Models with none (Midjourney, Flux Pro/Ultra, Imagen4, Ideogram, Recraft, Higgsfield Soul, Krea, Dreamina, and others) silently ignore source_images if passed here instead of erroring — if unsure, confirm the model appears in `list_models type="image_editing"` before trusting a bare id, or just use a known-safe default: "nano-banana-pro/edit" (best general prompt editor), "gpt-image-2" (photoreal, strong text), or "flux-2/edit".'),
171
189
  source_images: z.array(z.string()).describe('PIXEL-ACCURATE compositing. Array of source image URLs whose pixel content is composited into the output. **Cap: pass at most `max_reference_images` URLs from list_models for the chosen model — exceeding it is a deterministic 400.** Three modes the model auto-detects from input shape: (1) Single image → edit/transform that image. (2) Multiple images, one base + others → composite the others into the base. (3) Multiple images with no clear base → generate a new scene that pixel-accurately embeds the supplied images at positions described in the prompt. Mode 3 is the canonical pattern for thumbnails / branded compositions where exact-pixel logo + face fidelity matter. Refer to source images in the prompt by ordinal position ("FIRST source image", "SECOND source image") or use @image1/@image2 tags. Add "composite AS-IS, do not redraw or restyle" to lock pixels.'),
172
190
  aspect_ratio: z.string().optional().describe('Output aspect ratio (e.g., "1:1", "16:9", "9:16"). Must be in the chosen model\'s `supported_aspect_ratios` from list_models. Default: "1:1"'),
173
191
  enhance_prompt: z.boolean().optional().describe('Enhance the prompt for better results. Default: false — only pass true if the user explicitly asks to enhance/improve the prompt.'),
@@ -189,7 +207,8 @@ function registerGenerateTools(server, client, options = {}) {
189
207
 
190
208
  if (ui()) return uiGenerating({
191
209
  tool: 'generate_image_edit', kind: 'image', gen, client, model, prompt,
192
- count: num_images, settings: { resolution, aspect_ratio },
210
+ count: num_images,
211
+ settings: imageSettings({ resolution, aspect_ratio, enhance_prompt, enable_web_search, visual_dna_ids, moodboard_id, cinematic }),
193
212
  reference_image: source_images?.[0]
194
213
  });
195
214
 
@@ -1434,13 +1453,13 @@ function registerGenerateTools(server, client, options = {}) {
1434
1453
  .describe('Output quality preset (e.g. "high", "standard"). Applies where the underlying model supports quality tiers.'),
1435
1454
 
1436
1455
  ai_optimize: z.boolean().optional()
1437
- .describe('Whether to let Kolbo AI enhance your prompt before sending to the model. Default: true. Set false to use your prompt exactly as written.'),
1456
+ .describe('Whether to let Kolbo AI enhance your prompt before sending to the model. Default: false your prompt reaches the model exactly as written. Only pass true if the user explicitly asks to enhance/improve the prompt.'),
1438
1457
 
1439
1458
  project_id: projectIdField
1440
1459
  },
1441
1460
  async ({
1442
1461
  image_url, operation, model, scale, aspect_ratio, skin_strength, prompt,
1443
- mask_image_url, additional_images, generate_all_angles, resolution, quality, ai_optimize,
1462
+ mask_image_url, additional_images, generate_all_angles, resolution, quality, ai_optimize = false,
1444
1463
  zoom_out_percentage, expand_left, expand_right, expand_top, expand_bottom,
1445
1464
  project_id
1446
1465
  }) => {