@kolbo/mcp 1.38.0 → 1.40.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kolbo/mcp",
3
- "version": "1.38.0",
3
+ "version": "1.40.0",
4
4
  "description": "Kolbo AI MCP Server - Generate images, videos, music, speech, and sound effects from Claude Code",
5
5
  "main": "src/index.js",
6
6
  "bin": {
@@ -1,7 +1,7 @@
1
1
  # AUTO-GENERATED — do not edit
2
2
 
3
3
  This skill/ tree is mirrored from kolbo-code (the single source of truth)
4
- by .github/workflows/sync-skill-to-plugin.yml — synced from kolbo-code@82f9072.
4
+ by .github/workflows/sync-skill-to-plugin.yml — synced from kolbo-code@d70e982.
5
5
 
6
6
  It is the skill that 'npx @kolbo/mcp install' deploys into the user's agent.
7
7
  To change it, edit packages/opencode/skills/kolbo/ in kolbo-code and push;
package/skill/SKILL.md CHANGED
@@ -1,5 +1,5 @@
1
1
  ---
2
- version: 0.7.1
2
+ version: 0.7.2
3
3
  name: kolbo
4
4
  description: |
5
5
  Generate, edit, or analyze creative media via the Kolbo AI MCP server:
@@ -275,9 +275,6 @@ Chat renders markdown natively. `![alt](url)` = inline image. `[label](url)` = l
275
275
 
276
276
  Avoid bare URL dumps and HTML `<table>` grids — canvas already provides a gallery.
277
277
 
278
- - **3D models (`generate_3d`)**: NEVER paste the raw GLB/FBX/OBJ/USDZ file URLs as a link list. The result renders as a card with a live 3D preview + Download buttons ("View in Canvas"). Just say the model is ready and describe it — the user grabs formats from the card.
279
- - **NEVER surface a raw provider domain** (e.g. `*.fal.media`, `replicate.delivery`) in chat or `.kolbo/production.md`. All Kolbo outputs are Kolbo-hosted; if you ever see a provider URL in a result, treat it as a bug — report it, don't echo it.
280
-
281
278
  **After `generate_creative_director` completes** — share results as individual URLs, one per scene. Do NOT create an HTML grid artifact.
282
279
 
283
280
  **Always** record every URL in `.kolbo/production.md` — see `references/workflows/production-log.md`.
package/skill/VERSION CHANGED
@@ -1 +1 @@
1
- 0.7.1
1
+ 0.7.2
package/src/apps/html.js CHANGED
@@ -27,6 +27,49 @@ ${body}
27
27
  // ---- shared widget runtime ----
28
28
  var KOLBO_LOGO_FALLBACK = ${JSON.stringify(KOLBO_LOGO_SVG)};
29
29
  var KOLBO_LOGO = ${JSON.stringify(KOLBO_LOGO_IMG)};
30
+ // ---- shared inline SVG icon set (currentColor, 1em; replaces emoji so glyphs
31
+ // render identically in every host iframe instead of tofu boxes) ----
32
+ function _svg(inner, o) {
33
+ o = o || {};
34
+ var f = o.fill ? o.fill : 'none';
35
+ var s = o.fill ? 'none' : 'currentColor';
36
+ return '<svg class="k-ic" viewBox="0 0 24 24" width="1em" height="1em" fill="' + f + '" stroke="' + s +
37
+ '" stroke-width="' + (o.w || 2) + '" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true">' + inner + '</svg>';
38
+ }
39
+ var ICONS = {
40
+ upload: _svg('<path d="M21 15v4a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2v-4"/><path d="M17 8l-5-5-5 5"/><path d="M12 3v12"/>'),
41
+ download: _svg('<path d="M21 15v4a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2v-4"/><path d="M7 10l5 5 5-5"/><path d="M12 15V3"/>'),
42
+ play: _svg('<path d="M8 5v14l11-7z"/>', { fill: 'currentColor', w: 1 }),
43
+ pause: _svg('<path d="M6 4h4v16H6zM14 4h4v16h-4z"/>', { fill: 'currentColor', w: 1 }),
44
+ check: _svg('<path d="M20 6 9 17l-5-5"/>', { w: 2.4 }),
45
+ x: _svg('<path d="M18 6 6 18M6 6l12 12"/>', { w: 2.4 }),
46
+ warn: _svg('<path d="M10.29 3.86 1.82 18a2 2 0 0 0 1.71 3h16.94a2 2 0 0 0 1.71-3L13.71 3.86a2 2 0 0 0-3.42 0z"/><path d="M12 9v4"/><path d="M12 17h.01"/>'),
47
+ retry: _svg('<path d="M21 2v6h-6"/><path d="M3 12a9 9 0 0 1 15-6.7L21 8"/><path d="M3 22v-6h6"/><path d="M21 12a9 9 0 0 1-15 6.7L3 16"/>'),
48
+ edit: _svg('<path d="M12 20h9"/><path d="M16.5 3.5a2.12 2.12 0 0 1 3 3L7 19l-4 1 1-4z"/>'),
49
+ open: _svg('<path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"/><path d="M15 3h6v6"/><path d="M10 14 21 3"/>'),
50
+ arrowRight: _svg('<path d="M5 12h14"/><path d="M12 5l7 7-7 7"/>'),
51
+ sparkle: _svg('<path d="M12 3l1.9 5.1L19 10l-5.1 1.9L12 17l-1.9-5.1L5 10z"/>', { fill: 'currentColor', w: 1 }),
52
+ clock: _svg('<circle cx="12" cy="12" r="9"/><path d="M12 7v5l3 2"/>'),
53
+ sound: _svg('<path d="M11 5 6 9H2v6h4l5 4z"/><path d="M15.5 8.5a5 5 0 0 1 0 7"/><path d="M19 5a9 9 0 0 1 0 14"/>'),
54
+ mic: _svg('<path d="M12 2a3 3 0 0 0-3 3v6a3 3 0 0 0 6 0V5a3 3 0 0 0-3-3z"/><path d="M19 10a7 7 0 0 1-14 0"/><path d="M12 19v3"/>'),
55
+ image: _svg('<rect x="3" y="3" width="18" height="18" rx="2"/><circle cx="8.5" cy="8.5" r="1.5"/><path d="M21 15l-5-5L5 21"/>'),
56
+ video: _svg('<path d="M23 7l-7 5 7 5z"/><rect x="1" y="5" width="15" height="14" rx="2"/>'),
57
+ audio: _svg('<path d="M9 18V5l12-2v13"/><circle cx="6" cy="18" r="3"/><circle cx="18" cy="16" r="3"/>'),
58
+ document: _svg('<path d="M14 2H6a2 2 0 0 0-2 2v16a2 2 0 0 0 2 2h12a2 2 0 0 0 2-2V8z"/><path d="M14 2v6h6"/><path d="M8 13h8"/><path d="M8 17h8"/>'),
59
+ cube: _svg('<path d="M21 8l-9-5-9 5 9 5z"/><path d="M3 8v8l9 5 9-5V8"/><path d="M12 13v8"/>'),
60
+ file: _svg('<path d="M14 2H6a2 2 0 0 0-2 2v16a2 2 0 0 0 2 2h12a2 2 0 0 0 2-2V8z"/><path d="M14 2v6h6"/>')
61
+ };
62
+ // Media-kind → icon (accepts model kind / media_type strings).
63
+ function kindIcon(kind) {
64
+ switch (String(kind || '').toLowerCase()) {
65
+ case 'image': return ICONS.image;
66
+ case 'video': return ICONS.video;
67
+ case 'audio': case 'music': case 'sound': case 'speech': return ICONS.audio;
68
+ case 'document': case 'doc': return ICONS.document;
69
+ case '3d': case 'three_d': case 'model': return ICONS.cube;
70
+ default: return ICONS.file;
71
+ }
72
+ }
30
73
  function esc(s) {
31
74
  return String(s == null ? '' : s).replace(/[&<>"']/g, function (c) {
32
75
  return { '&': '&amp;', '<': '&lt;', '>': '&gt;', '"': '&quot;', "'": '&#39;' }[c];
package/src/apps/theme.js CHANGED
@@ -222,6 +222,8 @@ html.k-fullscreen .k-actions { flex: none; padding-top: 8px; }
222
222
  .k-btn.ghost { background: transparent; box-shadow: none; border-color: transparent; color: var(--text-muted); }
223
223
  .k-btn.ghost:hover { color: var(--text); background: var(--surface-2); }
224
224
  .k-btn svg { width: 13px; height: 13px; }
225
+ /* Inline SVG icons (replace emoji) — align with text, inherit color, never shrink. */
226
+ .k-ic { width: 1em; height: 1em; vertical-align: -0.14em; flex: none; }
225
227
  .k-btn:disabled { opacity: 0.5; cursor: default; }
226
228
 
227
229
  /* ---- Inline prompt input (Animate / Edit flows) ---- */
@@ -38,7 +38,7 @@ const BODY = `
38
38
  <div class="k-prompt-row" id="prompt-row">
39
39
  <input class="k-input" id="action-input" placeholder="">
40
40
  <button class="k-btn primary" id="action-send">Send</button>
41
- <button class="k-btn ghost" id="action-cancel">✕</button>
41
+ <button class="k-btn ghost" id="action-cancel" aria-label="Cancel"></button>
42
42
  </div>
43
43
  <div class="k-actions" id="actions"></div>
44
44
  </div>
@@ -55,6 +55,7 @@ var selected = 0; // selected result index
55
55
  var pollTimer = null;
56
56
 
57
57
  el('logo').innerHTML = KOLBO_LOGO + '<span>Kolbo</span>';
58
+ el('action-cancel').innerHTML = ICONS.x;
58
59
  el('kolbo-link').onclick = function (e) { e.preventDefault(); window.kolbo.openLink('https://app.kolbo.ai'); };
59
60
 
60
61
  var TOOL_TITLES = {
@@ -85,11 +86,11 @@ function renderChips(sc) {
85
86
  var h = modelChipHTML(sc.model, sc.model_icon);
86
87
  var s = sc.settings || {};
87
88
  if (sc.kind) h += chip(iconFor(sc.kind) + ' ' + sc.kind);
88
- if (s.duration) h += chip(' ' + fmtDur(s.duration));
89
+ if (s.duration) h += chip(ICONS.clock + ' ' + fmtDur(s.duration));
89
90
  if (s.resolution) h += chip(esc(s.resolution));
90
91
  if (s.aspect_ratio) h += chip(esc(s.aspect_ratio));
91
- if (s.audio) h += chip('🔊 audio');
92
- if (s.voice) h += chip('🎤 ' + esc(s.voice));
92
+ if (s.audio) h += chip(ICONS.sound + ' audio');
93
+ if (s.voice) h += chip(ICONS.mic + ' ' + esc(s.voice));
93
94
  if (s.mode) h += chip(esc(s.mode));
94
95
  if (sc.count > 1) h += chip('×' + sc.count);
95
96
  if (sc.reference_image) h += '<img class="k-ref-thumb" src="' + esc(sc.reference_image) + '" alt="" title="Reference image" onerror="this.style.display=\\'none\\'">';
@@ -97,7 +98,13 @@ function renderChips(sc) {
97
98
  }
98
99
  function chip(inner) { return '<span class="k-chip">' + inner + '</span>'; }
99
100
  function iconFor(kind) {
100
- return { image: '🖼', video: '🎬', audio: '🎵', '3d': '🧊', scenes: '🎞' }[kind] || '✨';
101
+ switch (kind) {
102
+ case 'image': return ICONS.image;
103
+ case 'video': case 'scenes': return ICONS.video;
104
+ case 'audio': return ICONS.audio;
105
+ case '3d': return ICONS.cube;
106
+ default: return ICONS.sparkle;
107
+ }
101
108
  }
102
109
 
103
110
  /* ---------- generating ---------- */
@@ -247,7 +254,7 @@ function renderLinks(urls) {
247
254
  // Small hover download button attached to a media cell (per-item downloads —
248
255
  // batch grids and CD scenes have no single "current" url for the action row).
249
256
  function dlBtnHTML(u) {
250
- return '<button class="k-dl" data-dl="' + esc(u) + '" title="Download" aria-label="Download">⬇</button>';
257
+ return '<button class="k-dl" data-dl="' + esc(u) + '" title="Download" aria-label="Download">' + ICONS.download + '</button>';
251
258
  }
252
259
  function wireDlButtons(root) {
253
260
  Array.prototype.forEach.call((root || document).querySelectorAll('.k-dl[data-dl]'), function (b) {
@@ -280,7 +287,7 @@ function renderScenes(sc) {
280
287
  : '<img id="scene-main" src="' + esc(it.url) + '" alt="" style="cursor:zoom-in">';
281
288
  var thumbs = '<div class="k-thumbs">' + items.map(function (t, i) {
282
289
  var inner = t.type === 'video'
283
- ? '<span style="display:flex;align-items:center;justify-content:center;width:100%;height:100%;background:rgba(255,255,255,0.06);font-size:14px">▶</span>'
290
+ ? '<span style="display:flex;align-items:center;justify-content:center;width:100%;height:100%;background:rgba(255,255,255,0.06);font-size:22px">' + ICONS.play + '</span>'
284
291
  : '<img src="' + esc(t.url) + '" alt="" loading="lazy">';
285
292
  return '<div class="k-thumb' + (i === selected ? ' active' : '') + '" data-i="' + i + '" title="' + esc(t.label) + '">' + inner + '</div>';
286
293
  }).join('') + '</div>';
@@ -324,8 +331,8 @@ function renderError(msg) {
324
331
  clearTimeout(pollTimer);
325
332
 
326
333
  setPhaseChip('Failed', false);
327
- el('stage').innerHTML = '<div class="k-error">⚠ ' + esc(msg) + '</div>';
328
- el('actions').innerHTML = '<button class="k-btn" id="retry-btn">↻ Try Again</button>';
334
+ el('stage').innerHTML = '<div class="k-error">' + ICONS.warn + ' ' + esc(msg) + '</div>';
335
+ el('actions').innerHTML = '<button class="k-btn" id="retry-btn">' + ICONS.retry + ' Try Again</button>';
329
336
  el('retry-btn').onclick = function () {
330
337
  var what = (state && TOOL_TITLES[state.tool]) || 'generation';
331
338
  window.kolbo.sendMessage('Please retry that ' + what.toLowerCase() + ' — it failed with: ' + msg);
@@ -355,7 +362,7 @@ function applyFullscreen(on, exitHandler) {
355
362
  var c = el('phase-chip');
356
363
  if (on) {
357
364
  c.style.display = '';
358
- c.innerHTML = ' ' + esc('Exit');
365
+ c.innerHTML = ICONS.x + ' ' + esc('Exit');
359
366
  c.style.cursor = 'pointer';
360
367
  c.onclick = exitHandler || toggleFullscreen;
361
368
  } else {
@@ -382,18 +389,18 @@ function renderActions(sc) {
382
389
  var a = [];
383
390
  var hasSingleUrl = !!(sc.urls && sc.urls.length);
384
391
  if (sc.kind === 'image') {
385
- a.push('<button class="k-btn primary" id="btn-animate">🎬 Animate</button>');
386
- a.push('<button class="k-btn" id="btn-edit">✏️ Edit</button>');
392
+ a.push('<button class="k-btn primary" id="btn-animate">' + ICONS.video + ' Animate</button>');
393
+ a.push('<button class="k-btn" id="btn-edit">' + ICONS.edit + ' Edit</button>');
387
394
  }
388
395
  if (sc.kind === 'video') {
389
- a.push('<button class="k-btn primary" id="btn-download">⬇ Download</button>');
396
+ a.push('<button class="k-btn primary" id="btn-download">' + ICONS.download + ' Download</button>');
390
397
  } else if (hasSingleUrl) {
391
398
  // Scenes (Creative Director) have no single "current" url — per-item hover
392
399
  // download buttons cover them instead.
393
- a.push('<button class="k-btn" id="btn-download">⬇ Download</button>');
400
+ a.push('<button class="k-btn" id="btn-download">' + ICONS.download + ' Download</button>');
394
401
  }
395
- a.push('<button class="k-btn" id="btn-recreate">↻ Recreate</button>');
396
- a.push('<button class="k-btn ghost" id="btn-open">Open in Kolbo ↗</button>');
402
+ a.push('<button class="k-btn" id="btn-recreate">' + ICONS.retry + ' Recreate</button>');
403
+ a.push('<button class="k-btn ghost" id="btn-open">Open in Kolbo ' + ICONS.open + '</button>');
397
404
  el('actions').innerHTML = a.join('');
398
405
 
399
406
  bind('btn-download', function () { window.kolbo.openLink(downloadUrl(currentUrl())); });
@@ -405,7 +412,7 @@ function renderActions(sc) {
405
412
  '\\n(from the ' + (TOOL_TITLES[state.tool] || 'generation') + ' widget)');
406
413
  });
407
414
  bind('btn-animate', function () {
408
- openPromptRow('Describe the motion (optional — Smart Select picks the best video model)…', function (text) {
415
+ openPromptRow('Describe the motion (optional)…', function (text) {
409
416
  window.kolbo.sendMessage('Animate this image into a short video' +
410
417
  '\\n🎬 Reference image: ' + currentUrl() +
411
418
  '\\nModel: smart select — pick the best image-to-video model' +
@@ -65,8 +65,8 @@ function cellHTML(item, i) {
65
65
  var idx = state.items.indexOf(item);
66
66
  var media = item.thumbnail
67
67
  ? '<img src="' + esc(item.thumbnail) + '" loading="lazy" alt="">'
68
- : '<div style="display:flex;align-items:center;justify-content:center;height:100%;color:var(--text-faint);font-size:20px">' +
69
- ({ video: '🎬', '3d': '🧊' }[item.media_type] || '🖼') + '</div>';
68
+ : '<div style="display:flex;align-items:center;justify-content:center;height:100%;color:var(--text-faint);font-size:22px">' +
69
+ kindIcon(item.media_type) + '</div>';
70
70
  return '<div class="k-cell" data-i="' + idx + '">' +
71
71
  '<div class="k-cell-media">' + media + '</div>' +
72
72
  '<div class="k-cell-label">' + esc(item.title || '') + '</div>' +
@@ -81,7 +81,7 @@ function audioRowHTML(item) {
81
81
  '<div class="k-audio-meta"><div class="k-audio-title">' + esc(item.title || '') + '</div>' +
82
82
  '<div class="k-audio-sub">' + esc(item.subtitle || '') + '</div></div>' +
83
83
  (item.preview_audio || item.url
84
- ? '<button class="k-play" data-play="' + esc(item.preview_audio || item.url) + '">▶</button>' : '') +
84
+ ? '<button class="k-play" data-play="' + esc(item.preview_audio || item.url) + '">' + ICONS.play + '</button>' : '') +
85
85
  '<button class="k-btn" data-use="' + idx + '">Use</button></div>';
86
86
  }
87
87
 
@@ -96,13 +96,13 @@ function wire() {
96
96
  b.onclick = function (e) {
97
97
  e.stopPropagation();
98
98
  var url = b.getAttribute('data-play');
99
- if (playing && playing.src === url && !playing.paused) { playing.pause(); b.textContent = '▶'; return; }
99
+ if (playing && playing.src === url && !playing.paused) { playing.pause(); b.innerHTML = ICONS.play; return; }
100
100
  if (playing) playing.pause();
101
- Array.prototype.forEach.call(document.querySelectorAll('[data-play]'), function (x) { x.textContent = '▶'; });
101
+ Array.prototype.forEach.call(document.querySelectorAll('[data-play]'), function (x) { x.innerHTML = ICONS.play; });
102
102
  playing = new Audio(url);
103
103
  playing.play();
104
- b.textContent = '⏸';
105
- playing.onended = function () { b.textContent = '▶'; };
104
+ b.innerHTML = ICONS.pause;
105
+ playing.onended = function () { b.innerHTML = ICONS.play; };
106
106
  };
107
107
  });
108
108
  }
@@ -54,7 +54,7 @@ function boot(sc) {
54
54
  }
55
55
  if (sc.phase === 'failed') {
56
56
  el('phase-chip').style.display = 'none';
57
- el('stage').innerHTML = '<div class="k-error">⚠ ' + esc(sc.error || 'Transcription failed') + '</div>';
57
+ el('stage').innerHTML = '<div class="k-error">' + ICONS.warn + ' ' + esc(sc.error || 'Transcription failed') + '</div>';
58
58
  return;
59
59
  }
60
60
  el('phase-chip').style.display = '';
@@ -67,9 +67,9 @@ function boot(sc) {
67
67
  'background:var(--surface);border:1px solid var(--border);font-size:12.5px;color:var(--text-muted);white-space:pre-wrap">' +
68
68
  esc(sc.text || '(empty transcript)') + '</div>';
69
69
  var a = [];
70
- if (sc.srt_url) a.push('<button class="k-btn primary" data-url="' + esc(sc.srt_url) + '">⬇ SRT</button>');
71
- if (sc.word_by_word_srt_url) a.push('<button class="k-btn" data-url="' + esc(sc.word_by_word_srt_url) + '">⬇ Word-by-word SRT</button>');
72
- if (sc.txt_url) a.push('<button class="k-btn" data-url="' + esc(sc.txt_url) + '">⬇ TXT</button>');
70
+ if (sc.srt_url) a.push('<button class="k-btn primary" data-url="' + esc(sc.srt_url) + '">' + ICONS.download + ' SRT</button>');
71
+ if (sc.word_by_word_srt_url) a.push('<button class="k-btn" data-url="' + esc(sc.word_by_word_srt_url) + '">' + ICONS.download + ' Word-by-word SRT</button>');
72
+ if (sc.txt_url) a.push('<button class="k-btn" data-url="' + esc(sc.txt_url) + '">' + ICONS.download + ' TXT</button>');
73
73
  a.push('<button class="k-btn ghost" id="btn-copy">Copy text</button>');
74
74
  el('actions').innerHTML = a.join('');
75
75
  Array.prototype.forEach.call(el('actions').querySelectorAll('[data-url]'), function (b) {
@@ -77,7 +77,7 @@ function boot(sc) {
77
77
  });
78
78
  var copyBtn = el('btn-copy');
79
79
  if (copyBtn) copyBtn.onclick = function () {
80
- try { navigator.clipboard.writeText(state.text || ''); copyBtn.textContent = 'Copied '; } catch (e) {}
80
+ try { navigator.clipboard.writeText(state.text || ''); copyBtn.innerHTML = 'Copied ' + ICONS.check; } catch (e) {}
81
81
  };
82
82
  window.kolbo.notifySize();
83
83
  }
@@ -107,7 +107,7 @@ window.kolbo.onToolResult(function (result) {
107
107
  if (result.isError || /error|failed/i.test(txt)) {
108
108
  el('phase-chip').style.display = 'none';
109
109
  el('stage').classList.remove('k-empty');
110
- el('stage').innerHTML = '<div class="k-error">⚠ ' + esc((txt || 'Transcription failed').slice(0, 300)) + '</div>';
110
+ el('stage').innerHTML = '<div class="k-error">' + ICONS.warn + ' ' + esc((txt || 'Transcription failed').slice(0, 300)) + '</div>';
111
111
  window.kolbo.notifySize();
112
112
  return;
113
113
  }
@@ -33,7 +33,7 @@ const BODY = `
33
33
  </div>
34
34
  <div class="k-body">
35
35
  <div id="drop" style="border:1.5px dashed var(--border);border-radius:12px;padding:26px 16px;text-align:center;cursor:pointer;transition:border-color .15s,background .15s">
36
- <div style="font-size:22px;margin-bottom:6px">⬆</div>
36
+ <div id="drop-icon" style="font-size:26px;line-height:1;margin-bottom:8px;color:var(--text-muted)"></div>
37
37
  <div style="font-size:13px;font-weight:600">Click or drop files here</div>
38
38
  <div id="accept-hint" style="font-size:11.5px;color:var(--text-muted);margin-top:4px"></div>
39
39
  </div>
@@ -51,6 +51,7 @@ const BODY = `
51
51
 
52
52
  const SCRIPT = `
53
53
  el('logo').innerHTML = KOLBO_LOGO + '<span>Kolbo</span>';
54
+ el('drop-icon').innerHTML = ICONS.upload;
54
55
  el('kolbo-link').onclick = function (e) { e.preventDefault(); window.kolbo.openLink('https://app.kolbo.ai/media-library'); };
55
56
 
56
57
  var state = null;
@@ -61,10 +62,10 @@ var active = 0;
61
62
  var sent = false;
62
63
 
63
64
  var KINDS = {
64
- image: { exts: ['jpg','jpeg','png','webp','gif','heic','heif','avif','bmp','tif','tiff'], icon: '🖼' },
65
- video: { exts: ['mp4','mov','webm','m4v','mkv','avi'], icon: '🎬' },
66
- audio: { exts: ['mp3','wav','m4a','aac','ogg','flac'], icon: '🎵' },
67
- document: { exts: ['pdf','txt','md','csv','json','docx','xlsx','pptx','doc','xls'], icon: '📄' }
65
+ image: { exts: ['jpg','jpeg','png','webp','gif','heic','heif','avif','bmp','tif','tiff'], icon: ICONS.image },
66
+ video: { exts: ['mp4','mov','webm','m4v','mkv','avi'], icon: ICONS.video },
67
+ audio: { exts: ['mp3','wav','m4a','aac','ogg','flac'], icon: ICONS.audio },
68
+ document: { exts: ['pdf','txt','md','csv','json','docx','xlsx','pptx','doc','xls'], icon: ICONS.document }
68
69
  };
69
70
 
70
71
  function classify(name) {
@@ -98,7 +99,7 @@ function showExpired() {
98
99
  el('drop').style.pointerEvents = 'none';
99
100
  el('drop').style.opacity = '0.5';
100
101
  el('notice').style.display = '';
101
- el('notice').innerHTML = ' This upload window expired. Ask Claude to open a new upload widget.';
102
+ el('notice').innerHTML = ICONS.clock + ' This upload window expired. Ask Claude to open a new upload widget.';
102
103
  window.kolbo.notifySize();
103
104
  }
104
105
 
@@ -190,12 +191,12 @@ function upload(it) {
190
191
 
191
192
  // ---- rendering ----
192
193
  function rowHtml(it) {
193
- var icon = it.kind && KINDS[it.kind] ? KINDS[it.kind].icon : '📎';
194
+ var icon = it.kind && KINDS[it.kind] ? KINDS[it.kind].icon : ICONS.file;
194
195
  var right = '';
195
196
  if (it.status === 'queued') right = '<span style="color:var(--text-muted)">queued</span>';
196
197
  else if (it.status === 'uploading') right = '<span style="color:var(--text-muted)">' + it.pct + '%</span>';
197
- else if (it.status === 'done') right = '<span style="color:#4ade80">✓ uploaded</span>';
198
- else right = '<span class="k-error" style="padding:0;border:0;background:none">✕ ' + esc(it.err || 'failed') + '</span> <a href="#" data-retry="' + it.id + '" style="font-size:11px">retry</a>';
198
+ else if (it.status === 'done') right = '<span style="color:#4ade80">' + ICONS.check + ' uploaded</span>';
199
+ else right = '<span class="k-error" style="padding:0;border:0;background:none">' + ICONS.x + ' ' + esc(it.err || 'failed') + '</span> <a href="#" data-retry="' + it.id + '" style="font-size:11px">retry</a>';
199
200
  var bar = it.status === 'uploading'
200
201
  ? '<div style="height:3px;border-radius:2px;background:var(--surface);margin-top:5px;overflow:hidden"><div id="bar-' + it.id + '" style="height:100%;width:' + it.pct + '%;background:var(--accent,#7c6cff);transition:width .2s"></div></div>'
201
202
  : '';
@@ -241,7 +242,7 @@ function render() {
241
242
  sent = true;
242
243
  var lines = done.map(function (i, idx) { return (idx + 1) + '. ' + i.file.name + ' (' + i.kind + '): ' + i.url; });
243
244
  window.kolbo.sendMessage('I uploaded ' + done.length + ' file(s) to my Kolbo media library:\\n' + lines.join('\\n') + '\\nContinue with these files.');
244
- el('actions').innerHTML = '<span style="font-size:12px;color:var(--text-muted)">✓ Sent to Claude — continuing…</span>';
245
+ el('actions').innerHTML = '<span style="font-size:12px;color:var(--text-muted)">' + ICONS.check + ' Sent to Claude — continuing…</span>';
245
246
  window.kolbo.notifySize();
246
247
  };
247
248
  el('btn-more').onclick = function () { el('picker').click(); };
package/src/index.js CHANGED
@@ -119,7 +119,9 @@ function createServer(opts = {}) {
119
119
  '5. Written deliverables (plans, briefs, scripts, research summaries) can live in Kolbo too: author them as AI Docs with `create_doc` (project-scoped, editable in the app, shareable via `share_doc`).',
120
120
  '6. DIRECTOR / BATCH JOBS: generate_creative_director runs its scenes (image OR video) in parallel and only reports state="completed" once EVERY scene is terminal. Video batches can take many minutes. If the tool returns `_timed_out:true`, the batch is STILL RUNNING on the server — call `get_creative_director_status` with the returned generation_id and keep checking until state="completed" to collect all scene outputs. NEVER conclude a Director run failed and fall back to plain generate_image/generate_video without first checking status — doing so wastes the user\'s credits by paying twice. If scenes already carry image_urls/video_urls, they are done; do not regenerate.',
121
121
  '7. SESSION CONTINUITY: keep one workflow in ONE session. chat_send_message and the generation tools return a `session_id` — for follow-ups, refinements, retries, or additional steps on the SAME task/theme, pass that same `session_id` back on the next call instead of starting fresh. Only OMIT session_id (start a new session) when the user genuinely switches to an unrelated task. Do not open a new conversation/session for every message of the same workflow — it fragments the user\'s history and loses context.',
122
- '8. LOCAL FILES / CHAT ATTACHMENTS: remote MCP tools CANNOT read files the user attached to the chat. When a claude.ai (browser/mobile) user has a local image/video/audio/document to use as a generation input, IMMEDIATELY call `media_upload_widget` — an upload card appears in the chat, they upload, and stable Kolbo CDN URLs come back in a follow-up message. Never ask them to re-attach the file in chat and never invent a URL. On Claude Desktop/Code with filesystem access, use `upload_media` with the absolute local path instead.'
122
+ '8. LOCAL FILES / CHAT ATTACHMENTS: remote MCP tools CANNOT read files the user attached to the chat. When a claude.ai (browser/mobile) user has a local image/video/audio/document to use as a generation input, IMMEDIATELY call `media_upload_widget` — an upload card appears in the chat, they upload, and stable Kolbo CDN URLs come back in a follow-up message. Never ask them to re-attach the file in chat and never invent a URL. On Claude Desktop/Code with filesystem access, use `upload_media` with the absolute local path instead.',
123
+ '9. MODEL SELECTION: ALWAYS pass a specific `model` on every generation tool — do NOT omit it. Omitting falls back to "Smart Select" auto-routing, which we deliberately avoid because it hides the model choice from the user and often picks a generic default. Choose the model that best fits the task and the user\'s intent (quality, speed, style, capability). If you are unsure which model to use for a given type, call `list_models` with the matching `type` and pick the recommended/flagship one, then pass its `identifier`. Only use Smart Select (omit `model`) if the user EXPLICITLY asks you to auto-pick.',
124
+ '10. IMAGE EDITING: for ANY prompt-driven / content edit of an existing image — "make it night", changing scene/lighting/colors, adding/removing/replacing objects, restyling — use `generate_image_edit` (it runs on strong dedicated editing models, same as image generation). Do NOT use `edit_image` for content edits — `edit_image` is ONLY for mechanical enhancements (upscale, reframe, remove-background, skin retouch). Its `magic_edit` operation is deprecated in favor of `generate_image_edit`.'
123
125
  ].join('\n')
124
126
  });
125
127
 
@@ -52,7 +52,7 @@ function registerGenerateTools(server, client, options = {}) {
52
52
  'Generate image(s) from a text prompt using Kolbo AI. Supports Visual DNA profiles (for character/style/product consistency), moodboards (for style direction), reference images (for composition guidance), batch generation (num_images), and web-search grounding. For EDITING an existing image, use generate_image_edit instead. For a coordinated multi-scene set (storyboard, ad campaign), use generate_creative_director. Returns the final image URL(s) when complete.',
53
53
  {
54
54
  prompt: z.string().describe('Text description of the image to generate'),
55
- model: z.string().optional().describe('Model identifier. Use list_models type="text_to_img" to see options. Omit for Smart Select.'),
55
+ model: z.string().optional().describe('Model identifier — REQUIRED in practice: pick a specific model, do NOT omit (omitting = Smart Select auto-pick, which we avoid). Strong current defaults: "nano-banana-2" (versatile, text rendering, multilingual) or "gpt-image-2" (photoreal, infographics). Call list_models type="text_to_img" to see all options and pick per the user\'s intent.'),
56
56
  aspect_ratio: z.string().optional().describe('Aspect ratio (e.g., "1:1", "16:9", "9:16"). Must be a value present in the model\'s `supported_aspect_ratios` from list_models — pass an unsupported value and the API rejects. Default: "1:1"'),
57
57
  enhance_prompt: z.boolean().optional().describe('Enhance the prompt for better results. Default: true'),
58
58
  num_images: z.number().optional().describe('Number of images to generate in one call. Default: 1. Note: some models (Midjourney etc.) have a fixed `images_per_request` and ignore this — check list_models.'),
@@ -92,7 +92,7 @@ function registerGenerateTools(server, client, options = {}) {
92
92
  urls: result.result.urls,
93
93
  model: result.result.model,
94
94
  prompt_used: result.result.prompt_used,
95
- _followup_hint: 'If the user asks to edit/change/modify this image next, pass urls[0] to generate_image_edit (free-form edits) or edit_image (upscale/reframe/removebg/enhance_skin/magic_edit). Do NOT call generate_image again.'
95
+ _followup_hint: 'If the user asks to edit/change/modify this image next (scene, lighting, objects, style, color — any content edit), pass urls[0] to generate_image_edit. Use edit_image ONLY for mechanical ops (upscale/reframe/removebg/enhance_skin). Do NOT call generate_image again.'
96
96
  }, null, 2)
97
97
  }, ...images]
98
98
  };
@@ -102,10 +102,10 @@ function registerGenerateTools(server, client, options = {}) {
102
102
  // ─── generate_image_edit ──────────────────────────────────
103
103
  server.tool(
104
104
  'generate_image_edit',
105
- 'Edit or transform an existing image using AI. Provide the source image URL(s) in `source_images` and describe the edit in `prompt` (e.g., "remove the background", "change the car color to red", "add sunglasses to the person"). Supports Visual DNA profiles and moodboards for style-consistent edits. For creating a brand new image from scratch, use generate_image. Returns the edited image URL(s) when complete.',
105
+ 'THE tool for ANY prompt-driven / content edit of an existing image changing the scene ("make it night", "change the sky to sunset"), adding/removing/replacing objects, restyling, recoloring, compositing, or any "edit this image to…" request. This is the image-editing equivalent of generate_image and runs on strong dedicated editing models (nano-banana-2, gpt-image-2). Provide the source image URL(s) in `source_images` and the instruction in `prompt`. Supports Visual DNA profiles and moodboards for style-consistent edits. Do NOT use `edit_image` for these — that tool is only for mechanical enhancements (upscale/reframe/remove-background/skin). For a brand-new image from scratch, use generate_image. Returns the edited image URL(s) when complete.',
106
106
  {
107
107
  prompt: z.string().describe('Description of the edit to apply (e.g., "remove the background", "change the sky to sunset")'),
108
- model: z.string().optional().describe('Model identifier. Use list_models type="image_editing" to see options. Omit for Smart Select.'),
108
+ model: z.string().optional().describe('Model identifier — REQUIRED in practice: pick a specific editing model, do NOT omit (omitting = Smart Select auto-pick, which we avoid). Strong current default: "nano-banana-2" (best general prompt-driven editor) or "gpt-image-2" for photoreal edits. Call list_models type="image_editing" to see all options and pick per the user\'s intent.'),
109
109
  source_images: z.array(z.string()).describe('PIXEL-ACCURATE compositing. Array of source image URLs whose pixel content is composited into the output. **Cap: pass at most `max_reference_images` URLs from list_models for the chosen model — exceeding it is a deterministic 400.** Three modes the model auto-detects from input shape: (1) Single image → edit/transform that image. (2) Multiple images, one base + others → composite the others into the base. (3) Multiple images with no clear base → generate a new scene that pixel-accurately embeds the supplied images at positions described in the prompt. Mode 3 is the canonical pattern for thumbnails / branded compositions where exact-pixel logo + face fidelity matter. Refer to source images in the prompt by ordinal position ("FIRST source image", "SECOND source image") or use @image1/@image2 tags. Add "composite AS-IS, do not redraw or restyle" to lock pixels.'),
110
110
  aspect_ratio: z.string().optional().describe('Output aspect ratio (e.g., "1:1", "16:9", "9:16"). Must be in the chosen model\'s `supported_aspect_ratios` from list_models. Default: "1:1"'),
111
111
  enhance_prompt: z.boolean().optional().describe('Enhance the prompt for better results. Default: true'),
@@ -162,7 +162,7 @@ function registerGenerateTools(server, client, options = {}) {
162
162
  {
163
163
  prompt: z.string().describe('Creative brief or concept describing the full set of scenes to generate'),
164
164
  scene_count: z.number().optional().describe('Number of scenes/images to generate, 1–8. Default: 4. Use this — NOT num_images — to control how many outputs are created.'),
165
- model: z.string().optional().describe('Model identifier applied to every scene. Omit for Smart Select.'),
165
+ model: z.string().optional().describe('Model identifier applied to every scene. Pick a SPECIFIC model — do NOT omit (omitting = Smart Select auto-pick, which we avoid); call list_models for this type and choose the model that best fits the user\'s intent.'),
166
166
  aspect_ratio: z.string().optional().describe('Aspect ratio applied to every scene (e.g., "1:1", "16:9", "9:16"). Must be in the chosen model\'s `supported_aspect_ratios` from list_models. Default: "1:1"'),
167
167
  workflow_type: z.string().optional().describe('"image" (default) or "video"'),
168
168
  duration: z.number().optional().describe('Duration in seconds per scene (video mode only). Must be a value in `supported_durations` from list_models, OR within `min_output_duration`-`max_output_duration`. E.g., 5 or 10.'),
@@ -598,7 +598,7 @@ function registerGenerateTools(server, client, options = {}) {
598
598
  'Generate a video from reference elements (images, videos, and/or audio) + a text prompt. Use when the user wants to animate specific uploaded/referenced assets — e.g. "animate this product", "put these 3 characters into a scene". IMPORTANT: different models accept different numbers of inputs — call list_models type="elements" and read elements_max_images / elements_max_videos / elements_max_audio on the chosen model before generating. For text-only → video use generate_video instead. For animating a single still image use generate_video_from_image. Returns the final video URL when complete.',
599
599
  {
600
600
  prompt: z.string().describe('Text description of the desired video / animation'),
601
- model: z.string().optional().describe('Model identifier. Use list_models type="elements" to see options (Seedance 2, Kling O3 Reference, Grok Imagine, Veo 3.1, etc.). Check elements_max_images / elements_max_videos / elements_max_audio on the model. Omit for Smart Select.'),
601
+ model: z.string().optional().describe('Model identifier. Use list_models type="elements" to see options (Seedance 2, Kling O3 Reference, Grok Imagine, Veo 3.1, etc.). Check elements_max_images / elements_max_videos / elements_max_audio on the model. Pick a SPECIFIC model — do NOT omit (omitting = Smart Select auto-pick, which we avoid); call list_models for this type and choose the model that best fits the user\'s intent.'),
602
602
  reference_images: z.array(z.string()).optional().describe('Array of public image URLs used as reference elements (product shots, character references, etc.). **Cap: pass at most `elements_max_images` URLs from list_models for the chosen model — exceeding it is a deterministic 400.**'),
603
603
  reference_videos: z.array(z.string()).optional().describe('Array of reference video URLs for models that accept video inputs. **Cap: pass at most `elements_max_videos` URLs from list_models — if the cap is 0 the model rejects videos.**'),
604
604
  audio_url: z.string().optional().describe('URL of a reference audio track. **Audio constraints: `elements_max_audio` from list_models gates whether audio is accepted at all; audio duration must fall within `min_audio_duration`-`max_audio_duration`; format must be in `supported_audio_formats` (if specified).**'),
@@ -681,7 +681,7 @@ function registerGenerateTools(server, client, options = {}) {
681
681
  first_frame: z.string().optional().describe('URL or absolute local path to the first frame (file mode — alternative to first_frame_url)'),
682
682
  last_frame: z.string().optional().describe('URL or absolute local path to the last frame (file mode — alternative to last_frame_url)'),
683
683
  prompt: z.string().optional().describe('Optional description of the desired motion between the two frames (e.g. "smooth camera dolly in")'),
684
- model: z.string().optional().describe('Model identifier. Use list_models type="firstlastgenerations" to see options. Omit for Smart Select.'),
684
+ model: z.string().optional().describe('Model identifier. Use list_models type="firstlastgenerations" to see options. Pick a SPECIFIC model — do NOT omit (omitting = Smart Select auto-pick, which we avoid); call list_models for this type and choose the model that best fits the user\'s intent.'),
685
685
  duration: z.number().optional().describe('Duration in seconds. Must be in `supported_durations` from list_models, OR within `min_output_duration`-`max_output_duration`. Default: 5'),
686
686
  aspect_ratio: z.string().optional().describe('Aspect ratio (auto-detected from first frame if not provided). Must be in `supported_aspect_ratios` from list_models when set. Default: "16:9"'),
687
687
  enhance_prompt: z.boolean().optional().describe('Enhance the prompt. Default: true'),
@@ -758,7 +758,7 @@ function registerGenerateTools(server, client, options = {}) {
758
758
  source: z.string().describe('URL or absolute local path to the source image or video (the face to animate). For lipsync-video: duration must fall within `min_video_duration`-`max_video_duration` from list_models.'),
759
759
  audio: z.string().describe('URL or absolute local path to the audio track (the voice to sync to). Duration must fall within `min_audio_duration`-`max_audio_duration` from list_models; format must be in `supported_audio_formats` (when set).'),
760
760
  text_prompt: z.string().optional().describe('Optional text prompt (for performance-capable models). For Sync-3 this is the free-text emotion/acting prompt, e.g. "speaking with excitement, calm and serious".'),
761
- model: z.string().optional().describe('Model identifier. Use list_models type="lipsync-image" or type="lipsync-video" to see options. Omit for Smart Select.'),
761
+ model: z.string().optional().describe('Model identifier. Use list_models type="lipsync-image" or type="lipsync-video" to see options. Pick a SPECIFIC model — do NOT omit (omitting = Smart Select auto-pick, which we avoid); call list_models for this type and choose the model that best fits the user\'s intent.'),
762
762
  bounding_box_target: z.array(z.number()).optional().describe('Optional bounding box [x, y, w, h] for multi-face inputs (Hedra Character3 style). Leave empty for single-face.'),
763
763
  // Sync-3 (fal-ai/sync-lipsync/v3) only; ignored by other models.
764
764
  sync_mode: z.enum(['cut_off', 'loop', 'bounce', 'silence', 'remap']).optional().describe('Sync-3 / sync-lipsync family: how to reconcile an audio/video length mismatch. Default cut_off.'),
@@ -867,7 +867,7 @@ function registerGenerateTools(server, client, options = {}) {
867
867
  {
868
868
  source_video: z.string().describe('URL or absolute local path to the primary source video to restyle. **Source duration must fall within `min_video_duration`-`max_video_duration` from list_models for the chosen model** — videos outside that range are rejected (or silently truncated by some upstream providers). For models that use reference_videos as their primary input (e.g. WAN 2.6 reference-to-video), pass the first reference video here and also include it in reference_videos.'),
869
869
  prompt: z.string().optional().describe('Text description of the desired restyle / transformation. Required by most video-to-video models; omit for prompt-less models (VEED Subtitles, Act Two, Wan Animate, Kling Motion Control).'),
870
- model: z.string().optional().describe('Model identifier. Use list_models type="video_to_video" to see options and check max_images / max_videos / max_elements / max_video_duration per model. Omit for Smart Select.'),
870
+ model: z.string().optional().describe('Model identifier. Use list_models type="video_to_video" to see options and check max_images / max_videos / max_elements / max_video_duration per model. Pick a SPECIFIC model — do NOT omit (omitting = Smart Select auto-pick, which we avoid); call list_models for this type and choose the model that best fits the user\'s intent.'),
871
871
  aspect_ratio: z.string().optional().describe('Output aspect ratio. Must be in `supported_aspect_ratios` from list_models when set. Default: matches source'),
872
872
  duration: z.number().optional().describe('Output duration in seconds. Must be in `supported_durations` from list_models, OR within `min_output_duration`-`max_output_duration`. Default: matches source'),
873
873
  enhance_prompt: z.boolean().optional().describe('Enhance the prompt. Default: true'),
@@ -1081,17 +1081,17 @@ function registerGenerateTools(server, client, options = {}) {
1081
1081
  // ─── edit_image ────────────────────────────────────────────
1082
1082
  server.tool(
1083
1083
  'edit_image',
1084
- 'Apply a targeted AI edit to an existing image. Use for upscaling resolution, changing aspect ratio (reframe), removing the background, portrait skin enhancement, or a text-guided edit (magic_edit). Faster and cheaper than generate_image_edit for these specific operations because it routes to specialized models. Returns the edited image URL when complete.',
1084
+ 'Apply a MECHANICAL enhancement to an existing image: upscale resolution, reframe (change aspect ratio), remove background, or portrait skin retouching. ⚠️ For any PROMPT-DRIVEN / CONTENT edit — changing the scene, lighting, or time of day, adding/removing/replacing objects, restyling, recoloring — use `generate_image_edit` instead (stronger dedicated editing models, better results). Do NOT use this tool for "make it night / add sunglasses / change the background to X"-type edits. Returns the edited image URL when complete.',
1085
1085
  {
1086
1086
  image_url: z.string().describe('URL of the source image to edit'),
1087
1087
  operation: z.enum(['upscale', 'reframe', 'removebg', 'enhance_skin', 'magic_edit'])
1088
- .describe('Edit operation to apply: "upscale" (increase resolution 2×–4×), "reframe" (change aspect ratio), "removebg" (remove background), "enhance_skin" (portrait retouching), "magic_edit" (text-guided edit — requires prompt)'),
1088
+ .describe('Mechanical edit operation: "upscale" (increase resolution 2×–4×), "reframe" (change aspect ratio), "removebg" (remove background), "enhance_skin" (portrait retouching). NOTE: "magic_edit" (text-guided content edit) is DEPRECATED here use `generate_image_edit` for prompt-driven edits instead; it produces better results on stronger models.'),
1089
1089
  model: z.string().optional().describe('Model identifier override. Omit to use the default model for the operation.'),
1090
1090
  scale: z.number().optional().describe('Upscale factor: 2, 3, or 4. Only used when operation="upscale". Default: 2.'),
1091
1091
  aspect_ratio: z.string().optional().describe('Target aspect ratio (e.g., "16:9", "9:16", "1:1"). Required for operation="reframe".'),
1092
1092
  skin_strength: z.enum(['subtle', 'realistic', 'pimple', 'freckle']).optional()
1093
1093
  .describe('Skin enhancement style. Only used when operation="enhance_skin". Default: "realistic".'),
1094
- prompt: z.string().optional().describe('Text instruction for the edit. Required for operation="magic_edit" (e.g., "add sunglasses", "change the sky to sunset").'),
1094
+ prompt: z.string().optional().describe('Text instruction — only for the deprecated operation="magic_edit". Prefer `generate_image_edit` for prompt-driven edits.'),
1095
1095
  project_id: projectIdField
1096
1096
  },
1097
1097
  async ({ image_url, operation, model, scale, aspect_ratio, skin_strength, prompt, project_id }) => {