@kolbo/mcp 1.88.1 → 1.88.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kolbo/mcp",
3
- "version": "1.88.1",
3
+ "version": "1.88.3",
4
4
  "description": "Kolbo AI MCP Server - Generate images, videos, music, speech, and sound effects from Claude Code",
5
5
  "main": "src/index.js",
6
6
  "bin": {
@@ -1,6 +1,6 @@
1
1
  # AUTO-GENERATED — do not edit
2
2
 
3
- This tree is mirrored from kolbo-code@06fbd86, the single source of truth.
3
+ This tree is mirrored from kolbo-code@0ac4fba, the single source of truth.
4
4
  Canonical source: packages/opencode/skills/kolbo/
5
5
  Distribution: .github/workflows/sync-skill-to-plugin.yml
6
6
 
package/skill/SKILL.md CHANGED
@@ -1,5 +1,5 @@
1
1
  ---
2
- version: 0.9.14
2
+ version: 0.9.15
3
3
  name: kolbo
4
4
  description: |
5
5
  Generate, edit, analyze, and direct creative media through Kolbo AI: images,
package/skill/VERSION CHANGED
@@ -1 +1 @@
1
- 0.9.14
1
+ 0.9.15
@@ -34,6 +34,31 @@ The backend renders internal specimens. Do NOT render specimens, attach them as
34
34
 
35
35
  Font upload/preparation has no separate credit charge; normal image generation remains billable under the existing approval rules. No automatic paid regeneration to improve typography.
36
36
 
37
+ ## Making the typeface actually come through
38
+
39
+ Measured 2026-09-10 by rerunning one customer ad (Hebrew, custom family, three reference
40
+ images) across models and settings. The backend renders a specimen and the model IMITATES
41
+ it — nothing installs the font — so these are the levers that decide how close it lands.
42
+
43
+ - **Model choice is the biggest one.** GPT Image 2 reproduced the uploaded letterforms
44
+ clearly better than GPT Image 2.5 Sunburst / Flare, which drift toward a default bold
45
+ Hebrew. Recommend GPT Image 2 whenever the typeface matters.
46
+ - **Quality does not compensate.** 2K + `high` on GPT Image 2 beat both 2.5 rows at
47
+ `max`. Do not sell a higher tier as a fix for typography.
48
+ - **Weight words in the prompt beat the specimen.** "bold", "medium weight", "very large
49
+ bold headline" read as typeface instructions and usually win — an ad that said bold five
50
+ times came back in a generic sans. Coach the user to describe size, placement, colour and
51
+ glow, and to choose the weight by selecting the uploaded STYLE (Bold / Medium / Light)
52
+ instead of writing it. Keep their exact-copy line ("EXACTLY letter for letter").
53
+ - **Busy layouts drift; calm ones do not.** The same font on a simple prompt reproduced
54
+ almost exactly, and on a split-screen ad with three competing references it was ignored.
55
+ Fewer competing reference images and fewer text blocks buy real fidelity.
56
+ - **Emoji never block a generation** and are drawn from the platform emoji set; they are
57
+ excluded from the specimen by design. No font carries them.
58
+ - **When it must be exact, say so.** For client-final work where the typeface cannot drift,
59
+ generate the layout with the text areas empty and set the type over it. Never promise
60
+ faithful reproduction — the model is imitating a picture of the letters.
61
+
37
62
  ## SDK / REST
38
63
 
39
64
  The account-authenticated server-side SDK exports `createFontClient`: `list`, `get`, `upload(Blob, filename)`, `status`, `rename`, `delete`, `createUploadTicket`, and `grantToApp`. Keep account API keys on the server. The dedicated REST root is `/api/v1/fonts`; multipart upload is POST to that root. App end-user credentials do not grant access to an owner's personal library; use explicit app font grants. Image SDK calls use the same optional `font_ids`.
@@ -1330,6 +1330,64 @@ function openPromptRow(placeholder, onSend) {
1330
1330
  The host mounts this iframe as soon as the tool is CALLED; the result can
1331
1331
  take many seconds (model resolution, file upload, submit). Show a live
1332
1332
  shell immediately instead of a blank card. */
1333
+ // Which tool args carry a reference the browser can actually load. The kind
1334
+ // here is only a FALLBACK: refKind() still lets the file extension win, exactly
1335
+ // like the result path, so a .mp4 handed to the files arg renders as video.
1336
+ var PRE_IMAGE_KEYS = ['source_images', 'reference_images', 'image_url', 'mask_image_url',
1337
+ 'additional_images', 'first_frame', 'last_frame', 'seed_reference_image_url',
1338
+ 'elements', 'files', 'keyframes', 'source'];
1339
+ var PRE_VIDEO_KEYS = ['source_video', 'reference_videos'];
1340
+ var PRE_AUDIO_KEYS = ['audio', 'audio_url', 'reference_audio_urls', 'seed_reference_audio_urls'];
1341
+
1342
+ // The card mounts the moment the tool is CALLED, so the only thing it knows is
1343
+ // the raw tool input - and for an edit/elements call the submit that follows is
1344
+ // the LONGEST wait on the card (local files are re-hosted first). Map the input
1345
+ // onto the same shape renderChips already reads for a server payload so the
1346
+ // references, DNA count and settings are on screen immediately instead of after
1347
+ // a minute of blank skeleton.
1348
+ // Deliberately NOT shown here: model, voice, DNA and moodboard NAMES. Those are
1349
+ // resolved server-side and arrive with the result (which overwrites all of
1350
+ // this) - rendering the raw identifier the caller passed would put an id on the
1351
+ // card, which is never allowed.
1352
+ function preRefSc(toolName, a) {
1353
+ var img = [], vid = [], aud = [];
1354
+ var take = function (v, bucket) {
1355
+ if (typeof v === 'string') {
1356
+ // http(s) only - an absolute local path is not loadable from the iframe.
1357
+ if (/^https?:/i.test(v) && bucket.indexOf(v) < 0) bucket.push(v);
1358
+ return;
1359
+ }
1360
+ if (Array.isArray(v)) {
1361
+ v.forEach(function (item) {
1362
+ take(typeof item === 'string' ? item : (item && (item.image_url || item.url)), bucket);
1363
+ });
1364
+ }
1365
+ };
1366
+ PRE_IMAGE_KEYS.forEach(function (k) { take(a[k], img); });
1367
+ PRE_VIDEO_KEYS.forEach(function (k) { take(a[k], vid); });
1368
+ PRE_AUDIO_KEYS.forEach(function (k) { take(a[k], aud); });
1369
+ return {
1370
+ tool: toolName,
1371
+ kind: kindFromTool(toolName, null),
1372
+ count: a.num_images || (Array.isArray(a.prompts) ? a.prompts.length : 1),
1373
+ reference_images: img,
1374
+ reference_videos: vid,
1375
+ reference_audio: aud,
1376
+ settings: {
1377
+ duration: a.duration,
1378
+ resolution: a.resolution,
1379
+ aspect_ratio: a.aspect_ratio,
1380
+ quality: a.quality,
1381
+ mode: a.mode,
1382
+ cinematic: a.cinematic,
1383
+ visual_dna_ids: a.visual_dna_ids,
1384
+ moodboard_ids: a.moodboard_ids,
1385
+ moodboard_id: a.moodboard_id,
1386
+ preset_id: a.preset_id
1387
+ }
1388
+ };
1389
+ }
1390
+
1333
1391
  function bootPre(toolName, args) {
1334
1392
  if (toolName) originTool = toolName;
1335
1393
  if (args) originArgs = args;
@@ -1347,6 +1405,7 @@ function bootPre(toolName, args) {
1347
1405
  setPrompt(promptHTML(raw), raw);
1348
1406
  }
1349
1407
  setPhaseChip('Preparing', true);
1408
+ renderChips(preRefSc(toolName, args || {}));
1350
1409
  if (!el('stage').innerHTML) {
1351
1410
  el('stage').innerHTML = '<div class="k-gen-grid n1"><div class="k-skel video" style="min-height:100px;max-height:140px"></div></div>';
1352
1411
  }