@kolbo/mcp 1.87.12 → 1.87.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,527 +1,531 @@
1
- /* ⛔ BACKWARD COMPATIBILITY: Tool names and arg names below are a PUBLIC
2
- * CONTRACT. Never rename, remove, or break an existing tool/arg — old cached
3
- * `npx @kolbo/mcp` installs in the wild will break silently. Add new tools or
4
- * new OPTIONAL args only. Full rules: ../index.js top-of-file and CLAUDE.md. */
5
-
6
- const { z } = require('zod');
7
- const { UI, uiResult, appsEnabled, resolveAvatarUrl } = require('../apps');
8
-
9
- // type name → human group label for the catalog widget
10
- const TYPE_GROUPS = {
11
- text_to_img: 'Image Generation',
12
- text_to_video: 'Video Generation',
13
- img_to_video: 'Video Generation',
14
- music_gen: 'Music',
15
- text_to_speech: 'Voice',
16
- image_editing: 'Image Editing',
17
- video_to_video: 'Video to Video',
18
- image_upscale: 'Image Upscale',
19
- image_reframe: 'Image Reframe',
20
- image_zoom_out: 'Image Expand',
21
- video_upscale: 'Video Upscale',
22
- video_reframe: 'Video Reframe',
23
- video_background_removal: 'Video Background Removal',
24
- video_to_sound: 'Video Audio Generation',
25
- video_face_swap: 'Video Face Swap',
26
- video_watermark_removal: 'Video Watermark Removal',
27
- video_extend: 'Video Extend',
28
- video_inpaint: 'Video Inpaint',
29
- video_retake: 'Video Retake',
30
- elements: 'Elements',
31
- };
32
-
33
- function groupNameFor(m) {
34
- const t = (Array.isArray(m.types) && m.types[0]) || m.type || '';
35
- if (TYPE_GROUPS[t]) return TYPE_GROUPS[t];
36
- if (t === 'three_d' || String(t).startsWith('3d_')) return '3D';
37
- return 'Other';
38
- }
39
-
40
- function modelChips(m) {
41
- const chips = [];
42
- if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
43
- const highest = [...m.supported_resolutions]
44
- .sort((a, b) => (parseInt(a, 10) || 0) - (parseInt(b, 10) || 0))
45
- .pop();
46
- if (highest) chips.push(String(highest));
47
- }
48
- if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
49
- const ds = [...m.supported_durations].sort((a, b) => a - b);
50
- chips.push(ds.length > 1 ? `${ds[0]}-${ds[ds.length - 1]}s` : `${ds[0]}s`);
51
- }
52
- if (m.supports_visual_dna) chips.push('DNA');
53
- if (m.new_model || m.newModel) chips.push('NEW');
54
- return chips.slice(0, 3);
55
- }
56
-
57
- // structuredContent for ui://kolbo/catalog.html — see src/apps/widgets/catalog.js
58
- // Deliberately CURATED, not exhaustive: the widget is a picker, not a database.
59
- // Each group shows the recommended/new models (max 6), and the total count
60
- // chip tells the user how many exist overall. Smart Select / "Auto" rows are
61
- // deliberately EXCLUDED — we always want a specific model chosen (server
62
- // instruction #9): auto-routing hides the model choice and the generation
63
- // metadata used to read just "Auto".
64
- function buildCatalogStructured(models, type, compact) {
65
- const groups = [];
66
- const byName = new Map();
67
- const isAuto = (m) => /^auto$|smart.select/i.test(String(m.name || '')) || /smart-select|k_auto/i.test(String(m.identifier || ''));
68
-
69
- // Recommended + new models float to the top of each group.
70
- const ranked = [...models].filter((m) => !isAuto(m)).sort((a, b) => {
71
- const score = (m) => (m.recommended ? 2 : 0) + (m.new_model || m.newModel ? 1 : 0);
72
- return score(b) - score(a);
73
- });
74
-
75
- for (const m of ranked) {
76
- const name = groupNameFor(m);
77
- let g = byName.get(name);
78
- if (!g) { g = { name, models: [] }; byName.set(name, g); groups.push(g); }
79
- if (g.models.length >= 6) continue; // curated cap — full list lives in the text payload
80
- g.models.push({
81
- name: m.name,
82
- // The widget renders `name`; the AGENT reads the same rows (hosts hand it
83
- // structuredContent). Without the identifier the default call was a dead
84
- // end — it named six models and gave no way to pass any of them on.
85
- identifier: m.identifier,
86
- icon: resolveAvatarUrl(m.avatar),
87
- description: String(m.smartSelect_StrengthsSummary || m.summary || m.description || '').slice(0, 90),
88
- chips: modelChips(m),
89
- use_hint: `Generate with the "${m.name}" model — ask me what I want to create first.`,
90
- });
91
- }
92
- groups.sort((a, b) => (a.name === 'Other' ? 1 : b.name === 'Other' ? -1 : 0));
93
- return {
94
- widget: 'catalog',
95
- title: 'Kolbo AI Models' + (type ? ' — ' + type : ''),
96
- total_available: models.length,
97
- compact: compact === true,
98
- groups,
99
- };
100
- }
101
-
102
- // One row per model — every identifier, nothing else. ~90 bytes/model, so the
103
- // whole 400+ model catalog fits in a payload an agent can actually read.
104
- const identifierRow = (m) => ({
105
- identifier: m.identifier,
106
- name: m.name,
107
- types: m.types,
108
- credit: m.credit,
109
- ...(m.video_input_credit != null ? { video_input_credit: m.video_input_credit } : {}),
110
- ...(m.recommended ? { recommended: true } : {}),
111
- ...(m.new_model ? { new_model: true } : {}),
112
- });
113
-
114
- function registerModelTools(server, client, options = {}) {
115
- const ui = () => appsEnabled(server, options);
116
- // ─── list_models ───────────────────────────────────────────
117
- server.tool(
118
- 'list_models',
119
- 'List available AI models on Kolbo. Filter by `type` to narrow to a generation type, and pass `format: "json"` to enumerate the catalog with exact identifiers — `format: "json"` + `type` returns the full raw model documents (every constraint field, for programmatic comparison / cap validation before submitting a generation); `format: "json"` alone returns a compact index of EVERY model and its identifier. Default `format: "text"` returns the human-readable summary. NEVER guess a model identifier: call this tool. ⚠️ COST: video / firstlast / elements / motion_graphic / cast rates are normally per output second. If a model publishes `video_input_credit` and the request includes one or more input videos, use that alternate rate and bill nominal input seconds + nominal output seconds. Encoder padding within 0.15s of an integer snaps to it; larger fractions round up. A `flat_credit_by_resolution` model instead charges the flat tier regardless of duration. Every other model type (image, audio, 3D, per-token text) bills as its catalog fields state.',
120
- {
121
- type: z.string().optional().describe('Filter by DB type name. Generation: "text_to_img", "image_editing", "text_to_video", "img_to_video", "draw_to_video", "video_to_video", "elements", "firstlastgenerations", "lipsync-image", "lipsync-video", "music_gen", "text_to_speech", "text_to_sound", "stt", "text". Image-edit engines: "image_upscale", "image_reframe", "image_zoom_out", "inpaint", "erase", "face_swap", "background_remove", "background_replace", "skin_enhancer", "graphics_enhance". Video-edit engines: "video_upscale", "video_reframe", "video_background_removal", "video_to_sound", "video_face_swap", "video_watermark_removal", "video_extend", "video_inpaint", "video_retake". For edit_image/edit_video, query the operation-specific type and pass a CONCRETE returned identifier; never submit a kolbo_gateway_* row, because those are web-navigation aliases rather than AI engines. Legacy aliases also accepted: "image", "image_edit", "video", "video_from_image", "video_from_video", "music", "speech", "sound", "chat", "lipsync", "three_d", "first_last_frame", "transcription". Omit for all models.'),
122
- format: z.enum(['text', 'json']).optional().describe('Output format. "text" (default) returns a human-readable summary with the most-used caps. "json" is the source of truth for identifiers and caps: with `type` it returns the raw model documents from the API (identifier, credit, supported_durations, supported_resolutions, supported_aspect_ratios, max_reference_images, max_visual_dna, max_video_duration, …) for EVERY model of that type; without `type` it returns a compact index of every model in the catalog and its exact identifier. Use it whenever you need an identifier you have not seen listed, or must verify a cap before passing a value that might exceed a model-specific limit.'),
123
- display_catalog: z.boolean().optional().describe('Set true when the USER explicitly asked to see/browse the available models — the visual catalog opens expanded. Leave unset for internal lookups (verifying a model name, checking caps before a generation): the catalog stays collapsed to a single row the user can tap to browse.')
124
- },
125
- async ({ type, format, display_catalog }) => {
126
- // The tool DECLARATION always carries widget meta, so hosts that mount
127
- // from tools/list (Claude Code desktop) prepare an iframe on EVERY call.
128
- // Returning plain text for internal lookups left that iframe with no
129
- // data — a dead, empty "Widget from Kolbo list_models" shell. Always
130
- // ship structuredContent; `compact` tells the widget to render a single
131
- // "Browse models" row (expandable) instead of the full catalog, which is
132
- // what display_catalog was really asking for.
133
- const showCatalog = display_catalog === true;
134
- const path = type ? `/v1/models?type=${encodeURIComponent(type)}` : '/v1/models';
135
- const result = await client.get(path);
136
-
137
- // ⚠️ Hosts that mount this widget (claude.ai, Claude Code desktop) hand the
138
- // MODEL `structuredContent` and DROP `content[].text`. So every payload the
139
- // agent needs has to ride in structuredContent — shipping it as text only
140
- // makes it invisible. That is exactly how `format: "json"` came to return
141
- // the curated 6-per-group picker instead of the raw documents: v1.53.1
142
- // (406a51e) flipped `if (ui() && showCatalog)` → `if (ui())` on all three
143
- // return paths, so the widget payload started shadowing the real answer and
144
- // the other 43 text_to_video identifiers became undiscoverable by any MCP
145
- // call. On 2026-08-09 that cost a wrong-model generation (minimax-h3).
146
- // `extra` (json mode) carries the data as structured fields; without it the
147
- // full text payload is attached verbatim. The widget ignores both.
148
- // Always ship structuredContent, exactly like listResult() does and for
149
- // the same reason: the tool DECLARATION carries widget meta, so a host
150
- // mounts the catalog iframe on every call — including hosts that never
151
- // advertised MCP Apps (Kolbo Code). Gating the payload on ui() left that
152
- // iframe with nothing to render and it sat on "Loading..." forever.
153
- const respond = (text, extra) =>
154
- uiResult(UI.catalog, text, {
155
- ...buildCatalogStructured(result.models, type, !showCatalog),
156
- ...(extra || { text }),
157
- });
158
-
159
- // JSON mode — the authoritative shape; every constraint the agent might
160
- // need to validate a request lives here (durations, reference caps,
161
- // audio/video min/max, resolution multipliers, supports_* flags,
162
- // prompt-length limits, etc.).
163
- if (format === 'json') {
164
- // Raw documents once `type` narrows the set (~49 docs for a video type).
165
- // Unfiltered that is 400+ documents / hundreds of KB, so return the
166
- // complete IDENTIFIER INDEX instead: every model stays enumerable and
167
- // the full caps are one `type` away.
168
- const payload = type
169
- ? { count: result.count, models: result.models }
170
- : {
171
- count: result.count,
172
- models: result.models.map(identifierRow),
173
- note: 'Compact index — every model in the catalog and its exact identifier. Re-call with `type` for the full documents (all caps, credit costs, supported_* fields).',
174
- };
175
- return respond(JSON.stringify(payload, null, 2), payload);
176
- }
177
-
178
- // Split into auto-selectable (has summary) and named-only (no summary)
179
- const withSummary = result.models.filter(m => m.summary && m.summary.trim() !== '');
180
- const withoutSummary = result.models.filter(m => !m.summary || m.summary.trim() === '');
181
-
182
- // Format the per-model spec line. The agent NEEDS this — without it,
183
- // it has to guess `supported_resolutions`/`supported_durations` and
184
- // either invents values (then the API silently substitutes) or asks
185
- // the user to clarify what's only knowable from this list.
186
- //
187
- // Rendering rule: emit a line for EVERY known constraint that is
188
- // applicable for this model's type — even when the value is 0 / null.
189
- // Hiding "0 cap" lines used to mean the agent couldn't distinguish
190
- // "this model rejects DNA" (cap = 0) from "I don't know" (field
191
- // missing). Now an explicit `max_dna: 0 (DNA not supported)` says the
192
- // model says no, and absence means the API doesn't expose the field.
193
- const formatSpecs = m => {
194
- const parts = [];
195
- if (m.haveThinking && Array.isArray(m.thinkingLevels) && m.thinkingLevels.length) {
196
- parts.push(`thinking_level: ${m.thinkingLevels.map(level => level.id).join('/')} (default ${m.thinkingDefault})`);
197
- }
198
- const types = Array.isArray(m.types) ? m.types : [];
199
- const isVideoType = types.some(t =>
200
- ['text_to_video', 'img_to_video', 'video_to_video', 'elements',
201
- 'firstlastgenerations', 'lipsync-image', 'lipsync-video', 'draw_to_video'].includes(t)
202
- );
203
- const isElements = types.includes('elements');
204
- const isV2V = types.includes('video_to_video');
205
- const isLipsyncVideo = types.includes('lipsync-video');
206
- const isLipsyncImage = types.includes('lipsync-image');
207
- const isImageEdit = types.includes('image_editing');
208
- const isImage = types.includes('text_to_img') || isImageEdit;
209
-
210
- if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
211
- const mult = m.resolution_multipliers || {};
212
- parts.push(
213
- 'resolutions: ' +
214
- m.supported_resolutions
215
- .map(r => (mult[r] != null && mult[r] !== 1 ? `${r} (${mult[r]}×)` : r))
216
- .join(' · ')
217
- );
218
- }
219
-
220
- if (m.video_input_credit != null) {
221
- const vm = m.video_input_resolution_multipliers || {};
222
- const tiers = Object.keys(vm).length
223
- ? ' · ' + Object.entries(vm).map(([r, mult]) => `${r} (${mult}×)`).join(' · ')
224
- : '';
225
- parts.push(`video_input_price: ${m.video_input_credit} credits/combined-second${tiers} · bill nominal input + output seconds (≤0.15s encoder padding snaps)`);
226
- }
227
-
228
- // Output durations (video gen output, not source video)
229
- if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
230
- const ds = m.supported_durations;
231
- const sorted = [...ds].sort((a, b) => a - b);
232
- const isRange = sorted.length > 2 && sorted.every((v, i) => i === 0 || v - sorted[i - 1] === 1);
233
- parts.push(`durations: ${isRange ? `${sorted[0]}-${sorted[sorted.length - 1]}s` : sorted.join('/') + 's'}`);
234
- } else if (isVideoType && (m.min_output_duration != null || m.max_output_duration != null)) {
235
- parts.push(`duration_range: ${m.min_output_duration ?? '?'}-${m.max_output_duration ?? '?'}s${m.default_duration != null ? ` (default ${m.default_duration}s)` : ''}`);
236
- }
237
-
238
- // Aspect ratios — prefer per-type override if set
239
- const ratios = m.supported_aspect_ratios_by_type
240
- ? Object.entries(m.supported_aspect_ratios_by_type).map(([t, arr]) => `${t}: ${arr.join('/')}`)
241
- : null;
242
- if (ratios) {
243
- parts.push(`aspect (per-type): ${ratios.join(' | ')}`);
244
- } else if (Array.isArray(m.supported_aspect_ratios) && m.supported_aspect_ratios.length) {
245
- parts.push(`aspect: ${m.supported_aspect_ratios.join(', ')}${m.default_aspect_ratio ? ` (default ${m.default_aspect_ratio})` : ''}`);
246
- }
247
-
248
- // Reference-input caps — show the slot relevant for this model family.
249
- // The same conceptual "max reference images" lives under THREE field
250
- // names depending on the model type. Be explicit about which is which
251
- // so the agent reads the right one.
252
- if (isImage || isImageEdit) {
253
- parts.push(`max_reference_images: ${m.max_reference_images ?? 0}${(m.max_reference_images ?? 0) === 0 ? ' (no refs)' : ''}`);
254
- }
255
- if (isElements) {
256
- parts.push(`elements caps: imgs=${m.elements_max_images ?? 0} · vids=${m.elements_max_videos ?? 0} · audio=${m.elements_max_audio ?? 0}`);
257
- }
258
- if (isV2V) {
259
- parts.push(`v2v ref caps: imgs=${m.max_images ?? 0} · vids=${m.max_videos ?? 0} · elements=${m.max_elements ?? 0} · audio=${m.max_audio ?? 0}`);
260
- }
261
-
262
- // Visual DNA cap — always show for image / elements / video, even if 0.
263
- // Use the authoritative supports_visual_dna flag when available; fall
264
- // back to inferring from cap > 0 for older API responses.
265
- const dnaSupported = typeof m.supports_visual_dna === 'boolean'
266
- ? m.supports_visual_dna
267
- : (m.max_visual_dna ?? 0) > 0;
268
- if (isImage || isVideoType) {
269
- const cap = m.max_visual_dna;
270
- if (dnaSupported && cap != null && cap > 0) parts.push(`max_visual_dna: ${cap}`);
271
- else if (dnaSupported && cap == null) parts.push('visual_dna: supported (no cap published — confirm before passing >3)');
272
- else parts.push('visual_dna: not supported');
273
- }
274
-
275
- // Source-video duration constraints — only matter for tools that take
276
- // an INPUT video (lipsync-video, video_to_video).
277
- if (isLipsyncVideo || isV2V) {
278
- if (m.min_video_duration != null || m.max_video_duration != null) {
279
- parts.push(`source_video: ${m.min_video_duration ?? '?'}-${m.max_video_duration ?? '?'}s`);
280
- }
281
- }
282
-
283
- // Audio input — lipsync, elements, music-driven flows.
284
- if (m.max_audio_duration != null || m.min_audio_duration != null) {
285
- parts.push(`audio_input: ${m.min_audio_duration ?? '?'}-${m.max_audio_duration ?? '?'}s${m.audio_max_follows_video_duration ? ' (max follows video)' : ''}`);
286
- }
287
- if (Array.isArray(m.supported_audio_formats) && m.supported_audio_formats.length) {
288
- parts.push(`audio_formats: ${m.supported_audio_formats.join('/')}`);
289
- }
290
-
291
- // Native sound generation (video models that emit synced audio)
292
- if (m.sound_generation_type === 'native') {
293
- const mult = m.sound_credit_multiplier && m.sound_credit_multiplier !== 1
294
- ? ` (${m.sound_credit_multiplier}×)`
295
- : '';
296
- parts.push(`sound: native${mult}${m.sound_enabled_by_default ? ' on-by-default' : ''}`);
297
- } else if (m.sound_baked_in) {
298
- parts.push('sound: baked-in (always on; type=none hides the toggle — still real audio)');
299
- }
300
-
301
- // Prompt constraints
302
- if (m.requires_prompt === false) parts.push('prompt: optional');
303
- if (m.min_prompt_length != null || m.max_prompt_length != null) {
304
- parts.push(`prompt_length: ${m.min_prompt_length ?? 0}-${m.max_prompt_length ?? '∞'} chars`);
305
- }
306
-
307
- // Upload cap (when present)
308
- if (m.max_file_size != null) {
309
- const mb = Math.round(m.max_file_size / (1024 * 1024));
310
- parts.push(`max_file_size: ${mb}MB`);
311
- }
312
-
313
- // Images-per-request (Midjourney-style fixed-N output)
314
- if (m.images_per_request != null && m.images_per_request !== 1) {
315
- parts.push(`images_per_request: ${m.images_per_request}`);
316
- }
317
-
318
- // Quality tiers (image models that support quality selection)
319
- if (Array.isArray(m.supported_qualities) && m.supported_qualities.length) {
320
- const qMult = m.quality_multipliers || {};
321
- const qParts = m.supported_qualities.map(q =>
322
- qMult[q] && qMult[q] !== 1 ? `${q}(${qMult[q]}×)` : q
323
- );
324
- parts.push(`quality: ${qParts.join(' · ')}${m.default_quality ? ` (default ${m.default_quality})` : ''}`);
325
- }
326
-
327
- // Fixed-price override (some models charge a flat rate per resolution instead of per-second)
328
- if (m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object' && Object.keys(m.flat_credit_by_resolution).length) {
329
- const fp = Object.entries(m.flat_credit_by_resolution).map(([k, v]) => `${k}:${v}cr`).join(' · ');
330
- parts.push(`flat_price: ${fp}`);
331
- }
332
-
333
- // Estimated generation time (wall-clock at base settings)
334
- if (m.estimated_duration_seconds != null) {
335
- parts.push(`est_time: ~${m.estimated_duration_seconds}s`);
336
- }
337
-
338
- // NSFW flag
339
- if (m.nsfw_only) {
340
- parts.push('nsfw: required');
341
- }
342
-
343
- return parts.length ? `\n ${parts.join(' | ')}` : '';
344
- };
345
-
346
- // The FULL catalog with every spec line measured 140,590 chars — past what
347
- // hosts accept, on the one discovery tool the skill tells the model to call
348
- // when it is unsure. Unfiltered, emit the one-line form (enough to choose a
349
- // model); once `type` narrows it, the set is small enough for full specs.
350
- const detailed = !!type;
351
- // Summaries run to a paragraph each; across the whole catalog that alone
352
- // is most of the payload. Unfiltered, one clause is enough to choose by.
353
- const brief = (s) => {
354
- if (!s) return '';
355
- const flat = String(s).replace(/\s+/g, ' ').trim();
356
- return flat.length > 130 ? flat.slice(0, 127).trimEnd() + '…' : flat;
357
- };
358
- // Text models bill per token — the flat `credit` is not what the user pays,
359
- // so show the real per-1K rates when the API supplies them. Without this the
360
- // "cheapest model that fits" rule is unusable for chat.
361
- //
362
- // Video-type models are the same problem in a different shape: kolbo-api's
363
- // credit engine (credManagment.js) treats "charge per second of requested
364
- // duration" as the UNIVERSAL rule for any type in
365
- // [video, firstlast, elements, motion_graphic, cast] — not a per-model
366
- // exception, the default. So `credit: 9` on a model with duration 8 is
367
- // really 72 credits, and nothing in the catalog said so: an agent quoting
368
- // cost from the bare `credit` field alone is wrong by exactly the
369
- // requested duration, every time. `flat_credit_by_resolution` is the one
370
- // carve-out — those models are charged the flat rate regardless of
371
- // duration, so they're excluded here the same way credManagment.js
372
- // excludes them (resolveFlatCredit wins over the multiplier).
373
- const PER_SECOND_TYPES = ['video', 'firstlast', 'elements', 'motion_graphic', 'cast'];
374
- const isPerSecondVideo = m => {
375
- const types = Array.isArray(m.types) ? m.types : (m.type ? [m.type] : []);
376
- const billedPerSecond = types.some(t => PER_SECOND_TYPES.some(kw => String(t).includes(kw)));
377
- const hasFlatOverride = m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object'
378
- && Object.keys(m.flat_credit_by_resolution).length > 0;
379
- return billedPerSecond && !hasFlatOverride;
380
- };
381
- const cost = m => (m.output_token_rate != null
382
- ? `${m.input_token_rate ?? '?'}/${m.output_token_rate} credits per 1K tokens (in/out)`
383
- : isPerSecondVideo(m)
384
- ? `${m.credit} credits/output-second${m.video_input_credit != null ? `; ${m.video_input_credit} credits/combined-second with video input` : ''}`
385
- : `${m.credit} credits`);
386
- const formatModel = m =>
387
- `${m.identifier} (${m.name}) - ${cost(m)}${m.recommended ? ' [RECOMMENDED]' : ''}${m.new_model ? ' [NEW]' : ''}${m.summary ? ` — ${detailed ? m.summary : brief(m.summary)}` : ''}${detailed ? formatSpecs(m) : ''}`;
388
-
389
- const sections = [];
390
-
391
- if (!detailed) {
392
- // The catalog is ~428 models. Listing all of them is both far past the
393
- // text budget AND useless to choose from — so unfiltered, surface the
394
- // curated picks and make the model narrow by `type` for the rest. This
395
- // matches the connector rule of steering to a CONCRETE model.
396
- const picks = result.models.filter(m => m.recommended || m.new_model);
397
- if (picks.length) {
398
- sections.push(`Recommended & new (${picks.length}):\n${picks.map(formatModel).join('\n')}`);
399
- }
400
- const text = `Kolbo model catalog — ${result.count} models total.\n\n`
401
- + `${sections.join('\n\n')}\n\n`
402
- + 'This shortlist is BADGE-BASED (recommended/new) — it is not a recommendation to '
403
- + 'use the newest or biggest model. To pick properly, re-call with `type` and choose by '
404
- + 'each model\'s strengths summary, taking the cheapest one that covers the task. '
405
- + 'To see everything in a '
406
- + 'category (with per-model resolutions, durations, aspect ratios and reference-image '
407
- + 'caps), re-call with `type`:\n'
408
- + ' text_to_img · image_editing · text_to_video · img_to_video · video_to_video ·\n'
409
- + ' first_last_frame · elements · lipsync · music_gen · text_to_speech ·\n'
410
- + ' text_to_sound · stt · three_d · text\n\n'
411
- + 'Use the "identifier" value as the "model" parameter in generate tools. '
412
- + 'For EVERY model + its exact identifier, re-call with format: "json" (compact index of the '
413
- + 'whole catalog). Add `type` to that call for the full raw documents with all caps.';
414
- return respond(text);
415
- }
416
-
417
- if (withSummary.length > 0) {
418
- sections.push(`Auto-selectable models (${withSummary.length}) — If the user already named a model/family (this turn or earlier), use that family — do not cheapest-swap. Otherwise CHOOSE BY THE SUMMARY after each "—": match it to what the user asked for, then take the CHEAPEST model that fits. Credit cost, [NEW] and [RECOMMENDED] are not reasons to pick a model:\n${withSummary.map(formatModel).join('\n')}`);
419
- }
420
- if (withoutSummary.length > 0) {
421
- sections.push(`Named-only models (${withoutSummary.length}) — only use if the user explicitly requests by name:\n${withoutSummary.map(formatModel).join('\n')}`);
422
- }
423
-
424
- const text = `Available ${type} models (${result.count}):\n\n${sections.join('\n\n')}\n\nEvery ${type} model in the catalog is listed above — both sections together are the complete set. Use the "identifier" value as the "model" parameter in generate tools. For the raw documents (programmatic cap validation), re-call with format: "json".`;
425
- return respond(text);
426
- }
427
- );
428
-
429
- // ─── check_credits ─────────────────────────────────────────
430
- server.tool(
431
- 'check_credits',
432
- 'Check your remaining Kolbo credit balance.',
433
- {},
434
- async () => {
435
- const result = await client.get('/v1/account/credits');
436
-
437
- return {
438
- content: [{
439
- type: 'text',
440
- text: `Credit Balance:\n- Total: ${result.credits.total}\n- Plan credits: ${result.credits.plan_credits}\n- Credit pack: ${result.credits.credit_pack}\n- Redemption: ${result.credits.redemption}`
441
- }]
442
- };
443
- }
444
- );
445
-
446
- // ─── get_session_usage ─────────────────────────────────────
447
- // Real, multiplier-adjusted credit spend tagged with the caller's
448
- // X-Kolbo-Caller-Session-Id (set automatically by the parent process —
449
- // no need to pass it). Use this to give the user an honest "you've spent
450
- // X credits in this app session" instead of estimating from base credits.
451
- server.tool(
452
- 'get_session_usage',
453
- 'Fetch real, multiplier-adjusted credit spend for the current Kolbo Code app session. Use when the user asks "how much did I spend?" or before/after a large bulk job so you can quote actual cost (not an estimate from base credits). Returns total + per-tool breakdown + per-model breakdown + a recent list. The caller-session-id is forwarded automatically by the MCP HTTP client. ONLY works when running under Kolbo Code — on the claude.ai connector and other hosts there is no per-app session to scope to; use check_credits there instead.',
454
- {},
455
- async () => {
456
- // Session scoping needs KOLBO_CALLER_SESSION_ID, which only the Kolbo Code
457
- // parent process sets. The remote connector serves every user from one
458
- // process, so it is never present there — calling anyway just returns a 400
459
- // telling the user to reconfigure a process they do not control.
460
- if (!process.env.KOLBO_CALLER_SESSION_ID) {
461
- return {
462
- content: [{
463
- type: 'text',
464
- text: JSON.stringify({
465
- unavailable: 'Per-session usage is only tracked when running under Kolbo Code.',
466
- reason: 'This host does not scope tool calls to an app session, so there is no session to total.',
467
- use_instead: 'check_credits for the current balance, or the Usage page at https://app.kolbo.ai.'
468
- })
469
- }]
470
- };
471
- }
472
- try {
473
- const r = await client.get('/credit-usage/by-caller-session');
474
- // The endpoint returns { message, data: { total, count, by_tool, by_model, recent[] } }
475
- return {
476
- content: [{
477
- type: 'text',
478
- text: JSON.stringify(r.data || r, null, 2)
479
- }]
480
- };
481
- } catch (err) {
482
- // 400 from the endpoint means no caller-session-id was forwarded —
483
- // surface a clear hint instead of a generic API error.
484
- const hint = err?.status === 400
485
- ? 'No caller-session-id was forwarded. Ensure the parent process (Kolbo Code / desktop sidecar) sets KOLBO_CALLER_SESSION_ID in this MCP\'s env, or call again after at least one media generation has fired.'
486
- : err?.message || 'Failed to fetch session usage';
487
- return {
488
- content: [{ type: 'text', text: JSON.stringify({ error: hint }, null, 2) }]
489
- };
490
- }
491
- }
492
- );
493
-
494
- // ─── show_plans ──────────────────────────────────
495
- // The upgrade card. Also rendered automatically when a generation is refused
496
- // for credits — see insufficientCreditsResult() in _shared.js.
497
- server.tool(
498
- 'show_plans',
499
- 'Show the user their Kolbo credit balance, current plan, and the available upgrade plans / credit packs as an interactive card. Use when the user asks about pricing, plans, upgrading, or how to get more credits. Prices shown are live and promo-adjusted. The card links to app.kolbo.ai/pricing to complete a purchase — never quote prices from memory, and never claim to have made a purchase for them.',
500
- {},
501
- async () => {
502
- const data = await client.get('/v1/account/plans');
503
- const structured = {
504
- widget: 'plans',
505
- reason: 'requested',
506
- balance: data?.credits?.total,
507
- current_plan: data?.current_plan || null,
508
- plans: data?.plans || [],
509
- credit_packs: data?.credit_packs || [],
510
- pricing_url: data?.pricing_url || 'https://app.kolbo.ai/pricing',
511
- };
512
- // structuredContent SHADOWS the text on widget hosts, so everything the
513
- // agent needs to talk about pricing has to live in the object above.
514
- const text = JSON.stringify({
515
- credits: structured.balance,
516
- current_plan: structured.current_plan,
517
- plans: structured.plans,
518
- credit_packs: structured.credit_packs,
519
- pricing_url: structured.pricing_url,
520
- _hint: 'A plans card is rendered for the user. Summarise briefly; do NOT paste the price table. Purchases are completed by the user on the pricing page — you cannot buy on their behalf.',
521
- }, null, 2);
522
- return uiResult(UI.plans, text, structured);
523
- }
524
- );
525
- }
526
-
527
- module.exports = { registerModelTools };
1
+ /* ⛔ BACKWARD COMPATIBILITY: Tool names and arg names below are a PUBLIC
2
+ * CONTRACT. Never rename, remove, or break an existing tool/arg — old cached
3
+ * `npx @kolbo/mcp` installs in the wild will break silently. Add new tools or
4
+ * new OPTIONAL args only. Full rules: ../index.js top-of-file and CLAUDE.md. */
5
+
6
+ const { z } = require('zod');
7
+ const { UI, uiResult, appsEnabled, resolveAvatarUrl } = require('../apps');
8
+
9
+ // type name → human group label for the catalog widget
10
+ const TYPE_GROUPS = {
11
+ text_to_img: 'Image Generation',
12
+ text_to_video: 'Video Generation',
13
+ img_to_video: 'Video Generation',
14
+ music_gen: 'Music',
15
+ text_to_speech: 'Voice',
16
+ image_editing: 'Image Editing',
17
+ video_to_video: 'Video to Video',
18
+ image_upscale: 'Image Upscale',
19
+ image_reframe: 'Image Reframe',
20
+ image_zoom_out: 'Image Expand',
21
+ video_upscale: 'Video Upscale',
22
+ video_reframe: 'Video Reframe',
23
+ video_background_removal: 'Video Background Removal',
24
+ video_to_sound: 'Video Audio Generation',
25
+ video_face_swap: 'Video Face Swap',
26
+ video_watermark_removal: 'Video Watermark Removal',
27
+ video_extend: 'Video Extend',
28
+ video_inpaint: 'Video Inpaint',
29
+ video_retake: 'Video Retake',
30
+ elements: 'Elements',
31
+ };
32
+
33
+ function groupNameFor(m) {
34
+ const t = (Array.isArray(m.types) && m.types[0]) || m.type || '';
35
+ if (TYPE_GROUPS[t]) return TYPE_GROUPS[t];
36
+ if (t === 'three_d' || String(t).startsWith('3d_')) return '3D';
37
+ return 'Other';
38
+ }
39
+
40
+ function modelChips(m) {
41
+ const chips = [];
42
+ if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
43
+ const highest = [...m.supported_resolutions]
44
+ .sort((a, b) => (parseInt(a, 10) || 0) - (parseInt(b, 10) || 0))
45
+ .pop();
46
+ if (highest) chips.push(String(highest));
47
+ }
48
+ if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
49
+ const ds = [...m.supported_durations].sort((a, b) => a - b);
50
+ chips.push(ds.length > 1 ? `${ds[0]}-${ds[ds.length - 1]}s` : `${ds[0]}s`);
51
+ }
52
+ if (m.supports_visual_dna) chips.push('DNA');
53
+ if (m.new_model || m.newModel) chips.push('NEW');
54
+ return chips.slice(0, 3);
55
+ }
56
+
57
+ // structuredContent for ui://kolbo/catalog.html — see src/apps/widgets/catalog.js
58
+ // Deliberately CURATED, not exhaustive: the widget is a picker, not a database.
59
+ // Each group shows the recommended/new models (max 6), and the total count
60
+ // chip tells the user how many exist overall. Smart Select / "Auto" rows are
61
+ // deliberately EXCLUDED — we always want a specific model chosen (server
62
+ // instruction #9): auto-routing hides the model choice and the generation
63
+ // metadata used to read just "Auto".
64
+ function buildCatalogStructured(models, type, compact) {
65
+ const groups = [];
66
+ const byName = new Map();
67
+ const isAuto = (m) => /^auto$|smart.select/i.test(String(m.name || '')) || /smart-select|k_auto/i.test(String(m.identifier || ''));
68
+
69
+ // Recommended + new models float to the top of each group.
70
+ const ranked = [...models].filter((m) => !isAuto(m)).sort((a, b) => {
71
+ const score = (m) => (m.recommended ? 2 : 0) + (m.new_model || m.newModel ? 1 : 0);
72
+ return score(b) - score(a);
73
+ });
74
+
75
+ for (const m of ranked) {
76
+ const name = groupNameFor(m);
77
+ let g = byName.get(name);
78
+ if (!g) { g = { name, models: [] }; byName.set(name, g); groups.push(g); }
79
+ if (g.models.length >= 6) continue; // curated cap — full list lives in the text payload
80
+ g.models.push({
81
+ name: m.name,
82
+ // The widget renders `name`; the AGENT reads the same rows (hosts hand it
83
+ // structuredContent). Without the identifier the default call was a dead
84
+ // end — it named six models and gave no way to pass any of them on.
85
+ identifier: m.identifier,
86
+ icon: resolveAvatarUrl(m.avatar),
87
+ description: String(m.smartSelect_StrengthsSummary || m.summary || m.description || '').slice(0, 90),
88
+ chips: modelChips(m),
89
+ use_hint: `Generate with the "${m.name}" model — ask me what I want to create first.`,
90
+ });
91
+ }
92
+ groups.sort((a, b) => (a.name === 'Other' ? 1 : b.name === 'Other' ? -1 : 0));
93
+ return {
94
+ widget: 'catalog',
95
+ title: 'Kolbo AI Models' + (type ? ' — ' + type : ''),
96
+ total_available: models.length,
97
+ compact: compact === true,
98
+ groups,
99
+ };
100
+ }
101
+
102
+ // One row per model — every identifier, nothing else. ~90 bytes/model, so the
103
+ // whole 400+ model catalog fits in a payload an agent can actually read.
104
+ const identifierRow = (m) => ({
105
+ identifier: m.identifier,
106
+ name: m.name,
107
+ types: m.types,
108
+ credit: m.credit,
109
+ ...(m.video_input_credit != null ? { video_input_credit: m.video_input_credit } : {}),
110
+ ...(m.recommended ? { recommended: true } : {}),
111
+ ...(m.new_model ? { new_model: true } : {}),
112
+ });
113
+
114
+ function registerModelTools(server, client, options = {}) {
115
+ const ui = () => appsEnabled(server, options);
116
+ // ─── list_models ───────────────────────────────────────────
117
+ server.tool(
118
+ 'list_models',
119
+ 'List available AI models on Kolbo. Filter by `type` to narrow to a generation type, and pass `format: "json"` to enumerate the catalog with exact identifiers — `format: "json"` + `type` returns the full raw model documents (every constraint field, for programmatic comparison / cap validation before submitting a generation); `format: "json"` alone returns a compact index of EVERY model and its identifier. Default `format: "text"` returns the human-readable summary. NEVER guess a model identifier: call this tool. ⚠️ COST: video / firstlast / elements / motion_graphic / cast rates are normally per output second. If a model publishes `video_input_credit` and the request includes one or more input videos, use that alternate rate and bill nominal input seconds + nominal output seconds. Encoder padding within 0.15s of an integer snaps to it; larger fractions round up. A `flat_credit_by_resolution` model instead charges the flat tier regardless of duration. Every other model type (image, audio, 3D, per-token text) bills as its catalog fields state.',
120
+ {
121
+ type: z.string().optional().describe('Filter by DB type name. Generation: "text_to_img", "image_editing", "text_to_video", "img_to_video", "draw_to_video", "video_to_video", "elements", "firstlastgenerations", "lipsync-image", "lipsync-video", "music_gen", "text_to_speech", "text_to_sound", "stt", "text". Image-edit engines: "image_upscale", "image_reframe", "image_zoom_out", "inpaint", "erase", "face_swap", "background_remove", "background_replace", "skin_enhancer", "graphics_enhance". Video-edit engines: "video_upscale", "video_reframe", "video_background_removal", "video_to_sound", "video_face_swap", "video_watermark_removal", "video_extend", "video_inpaint", "video_retake". For edit_image/edit_video, query the operation-specific type and pass a CONCRETE returned identifier; never submit a kolbo_gateway_* row, because those are web-navigation aliases rather than AI engines. Legacy aliases also accepted: "image", "image_edit", "video", "video_from_image", "video_from_video", "music", "speech", "sound", "chat", "lipsync", "three_d", "first_last_frame", "transcription". Omit for all models.'),
122
+ format: z.enum(['text', 'json']).optional().describe('Output format. "text" (default) returns a human-readable summary with the most-used caps. "json" is the source of truth for identifiers and caps: with `type` it returns the raw model documents from the API (identifier, credit, supported_durations, supported_resolutions, supported_aspect_ratios, max_reference_images, max_visual_dna, max_video_duration, …) for EVERY model of that type; without `type` it returns a compact index of every model in the catalog and its exact identifier. Use it whenever you need an identifier you have not seen listed, or must verify a cap before passing a value that might exceed a model-specific limit.'),
123
+ display_catalog: z.boolean().optional().describe('Set true when the USER explicitly asked to see/browse the available models — the visual catalog opens expanded. Leave unset for internal lookups (verifying a model name, checking caps before a generation): the catalog stays collapsed to a single row the user can tap to browse.')
124
+ },
125
+ async ({ type, format, display_catalog }) => {
126
+ // The tool DECLARATION always carries widget meta, so hosts that mount
127
+ // from tools/list (Claude Code desktop) prepare an iframe on EVERY call.
128
+ // Returning plain text for internal lookups left that iframe with no
129
+ // data — a dead, empty "Widget from Kolbo list_models" shell. Always
130
+ // ship structuredContent; `compact` tells the widget to render a single
131
+ // "Browse models" row (expandable) instead of the full catalog, which is
132
+ // what display_catalog was really asking for.
133
+ const showCatalog = display_catalog === true;
134
+ const path = type ? `/v1/models?type=${encodeURIComponent(type)}` : '/v1/models';
135
+ const result = await client.get(path);
136
+
137
+ // ⚠️ Hosts that mount this widget (claude.ai, Claude Code desktop) hand the
138
+ // MODEL `structuredContent` and DROP `content[].text`. So every payload the
139
+ // agent needs has to ride in structuredContent — shipping it as text only
140
+ // makes it invisible. That is exactly how `format: "json"` came to return
141
+ // the curated 6-per-group picker instead of the raw documents: v1.53.1
142
+ // (406a51e) flipped `if (ui() && showCatalog)` → `if (ui())` on all three
143
+ // return paths, so the widget payload started shadowing the real answer and
144
+ // the other 43 text_to_video identifiers became undiscoverable by any MCP
145
+ // call. On 2026-08-09 that cost a wrong-model generation (minimax-h3).
146
+ // `extra` (json mode) carries the data as structured fields; without it the
147
+ // full text payload is attached verbatim. The widget ignores both.
148
+ // Always ship structuredContent, exactly like listResult() does and for
149
+ // the same reason: the tool DECLARATION carries widget meta, so a host
150
+ // mounts the catalog iframe on every call — including hosts that never
151
+ // advertised MCP Apps (Kolbo Code). Gating the payload on ui() left that
152
+ // iframe with nothing to render and it sat on "Loading..." forever.
153
+ const respond = (text, extra) =>
154
+ uiResult(UI.catalog, text, {
155
+ ...buildCatalogStructured(result.models, type, !showCatalog),
156
+ ...(extra || { text }),
157
+ });
158
+
159
+ // JSON mode — the authoritative shape; every constraint the agent might
160
+ // need to validate a request lives here (durations, reference caps,
161
+ // audio/video min/max, resolution multipliers, supports_* flags,
162
+ // prompt-length limits, etc.).
163
+ if (format === 'json') {
164
+ // Raw documents once `type` narrows the set (~49 docs for a video type).
165
+ // Unfiltered that is 400+ documents / hundreds of KB, so return the
166
+ // complete IDENTIFIER INDEX instead: every model stays enumerable and
167
+ // the full caps are one `type` away.
168
+ const payload = type
169
+ ? { count: result.count, models: result.models }
170
+ : {
171
+ count: result.count,
172
+ models: result.models.map(identifierRow),
173
+ note: 'Compact index — every model in the catalog and its exact identifier. Re-call with `type` for the full documents (all caps, credit costs, supported_* fields).',
174
+ };
175
+ return respond(JSON.stringify(payload, null, 2), payload);
176
+ }
177
+
178
+ // Split into auto-selectable (has summary) and named-only (no summary)
179
+ const withSummary = result.models.filter(m => m.summary && m.summary.trim() !== '');
180
+ const withoutSummary = result.models.filter(m => !m.summary || m.summary.trim() === '');
181
+
182
+ // Format the per-model spec line. The agent NEEDS this — without it,
183
+ // it has to guess `supported_resolutions`/`supported_durations` and
184
+ // either invents values (then the API silently substitutes) or asks
185
+ // the user to clarify what's only knowable from this list.
186
+ //
187
+ // Rendering rule: emit a line for EVERY known constraint that is
188
+ // applicable for this model's type — even when the value is 0 / null.
189
+ // Hiding "0 cap" lines used to mean the agent couldn't distinguish
190
+ // "this model rejects DNA" (cap = 0) from "I don't know" (field
191
+ // missing). Now an explicit `max_dna: 0 (DNA not supported)` says the
192
+ // model says no, and absence means the API doesn't expose the field.
193
+ const formatSpecs = m => {
194
+ const parts = [];
195
+ if (m.haveThinking && Array.isArray(m.thinkingLevels) && m.thinkingLevels.length) {
196
+ parts.push(`thinking_level: ${m.thinkingLevels.map(level => level.id).join('/')} (default ${m.thinkingDefault})`);
197
+ }
198
+ const types = Array.isArray(m.types) ? m.types : [];
199
+ const isVideoType = types.some(t =>
200
+ ['text_to_video', 'img_to_video', 'video_to_video', 'elements',
201
+ 'firstlastgenerations', 'lipsync-image', 'lipsync-video', 'draw_to_video'].includes(t)
202
+ );
203
+ const isElements = types.includes('elements');
204
+ const isV2V = types.includes('video_to_video');
205
+ const isLipsyncVideo = types.includes('lipsync-video');
206
+ const isLipsyncImage = types.includes('lipsync-image');
207
+ const isImageEdit = types.includes('image_editing');
208
+ const isImage = types.includes('text_to_img') || isImageEdit;
209
+
210
+ if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
211
+ const mult = m.resolution_multipliers || {};
212
+ parts.push(
213
+ 'resolutions: ' +
214
+ m.supported_resolutions
215
+ .map(r => (mult[r] != null && mult[r] !== 1 ? `${r} (${mult[r]}×)` : r))
216
+ .join(' · ')
217
+ );
218
+ }
219
+
220
+ if (m.video_input_credit != null) {
221
+ const vm = m.video_input_resolution_multipliers || {};
222
+ const tiers = Object.keys(vm).length
223
+ ? ' · ' + Object.entries(vm).map(([r, mult]) => `${r} (${mult}×)`).join(' · ')
224
+ : '';
225
+ parts.push(`video_input_price: ${m.video_input_credit} credits/combined-second${tiers} · bill nominal input + output seconds (≤0.15s encoder padding snaps)`);
226
+ }
227
+
228
+ // Output durations (video gen output, not source video)
229
+ if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
230
+ const ds = m.supported_durations;
231
+ const sorted = [...ds].sort((a, b) => a - b);
232
+ const isRange = sorted.length > 2 && sorted.every((v, i) => i === 0 || v - sorted[i - 1] === 1);
233
+ parts.push(`durations: ${isRange ? `${sorted[0]}-${sorted[sorted.length - 1]}s` : sorted.join('/') + 's'}`);
234
+ } else if (isVideoType && (m.min_output_duration != null || m.max_output_duration != null)) {
235
+ parts.push(`duration_range: ${m.min_output_duration ?? '?'}-${m.max_output_duration ?? '?'}s${m.default_duration != null ? ` (default ${m.default_duration}s)` : ''}`);
236
+ }
237
+
238
+ // Aspect ratios — prefer per-type override if set
239
+ const ratios = m.supported_aspect_ratios_by_type
240
+ ? Object.entries(m.supported_aspect_ratios_by_type).map(([t, arr]) => `${t}: ${arr.join('/')}`)
241
+ : null;
242
+ if (ratios) {
243
+ parts.push(`aspect (per-type): ${ratios.join(' | ')}`);
244
+ } else if (Array.isArray(m.supported_aspect_ratios) && m.supported_aspect_ratios.length) {
245
+ parts.push(`aspect: ${m.supported_aspect_ratios.join(', ')}${m.default_aspect_ratio ? ` (default ${m.default_aspect_ratio})` : ''}`);
246
+ }
247
+
248
+ // Reference-input caps — show the slot relevant for this model family.
249
+ // The same conceptual "max reference images" lives under THREE field
250
+ // names depending on the model type. Be explicit about which is which
251
+ // so the agent reads the right one.
252
+ if (isImage || isImageEdit) {
253
+ parts.push(`max_reference_images: ${m.max_reference_images ?? 0}${(m.max_reference_images ?? 0) === 0 ? ' (no refs)' : ''}`);
254
+ }
255
+ if (isElements) {
256
+ parts.push(`elements caps: imgs=${m.elements_max_images ?? 0} · vids=${m.elements_max_videos ?? 0} · audio=${m.elements_max_audio ?? 0}`);
257
+ }
258
+ if (isV2V) {
259
+ parts.push(`v2v ref caps: imgs=${m.max_images ?? 0} · vids=${m.max_videos ?? 0} · elements=${m.max_elements ?? 0} · audio=${m.max_audio ?? 0}`);
260
+ }
261
+
262
+ // Visual DNA cap — always show for image / elements / video, even if 0.
263
+ // Use the authoritative supports_visual_dna flag when available; fall
264
+ // back to inferring from cap > 0 for older API responses.
265
+ const dnaSupported = typeof m.supports_visual_dna === 'boolean'
266
+ ? m.supports_visual_dna
267
+ : (m.max_visual_dna ?? 0) > 0;
268
+ if (isImage || isVideoType) {
269
+ const cap = m.max_visual_dna;
270
+ if (dnaSupported && cap != null && cap > 0) parts.push(`max_visual_dna: ${cap}`);
271
+ else if (dnaSupported && cap == null) parts.push('visual_dna: supported (no cap published — confirm before passing >3)');
272
+ else parts.push('visual_dna: not supported');
273
+ }
274
+
275
+ // Source-video duration constraints — only matter for tools that take
276
+ // an INPUT video (lipsync-video, video_to_video).
277
+ if (isLipsyncVideo || isV2V) {
278
+ if (m.min_video_duration != null || m.max_video_duration != null) {
279
+ parts.push(`source_video: ${m.min_video_duration ?? '?'}-${m.max_video_duration ?? '?'}s`);
280
+ }
281
+ }
282
+
283
+ // Audio input — lipsync, elements, music-driven flows.
284
+ if (m.max_audio_duration != null || m.min_audio_duration != null) {
285
+ parts.push(`audio_input: ${m.min_audio_duration ?? '?'}-${m.max_audio_duration ?? '?'}s${m.audio_max_follows_video_duration ? ' (max follows video)' : ''}`);
286
+ }
287
+ if (Array.isArray(m.supported_audio_formats) && m.supported_audio_formats.length) {
288
+ parts.push(`audio_formats: ${m.supported_audio_formats.join('/')}`);
289
+ }
290
+
291
+ // Native sound generation (video models that emit synced audio)
292
+ if (m.sound_generation_type === 'native') {
293
+ const mult = m.sound_credit_multiplier && m.sound_credit_multiplier !== 1
294
+ ? ` (${m.sound_credit_multiplier}×)`
295
+ : '';
296
+ parts.push(`sound: native${mult}${m.sound_enabled_by_default ? ' on-by-default' : ''}`);
297
+ } else if (m.sound_baked_in) {
298
+ parts.push('sound: baked-in (always on; type=none hides the toggle — still real audio)');
299
+ }
300
+
301
+ // Prompt constraints
302
+ if (m.requires_prompt === false) parts.push('prompt: optional');
303
+ if (m.min_prompt_length != null || m.max_prompt_length != null) {
304
+ parts.push(`prompt_length: ${m.min_prompt_length ?? 0}-${m.max_prompt_length ?? '∞'} chars`);
305
+ }
306
+
307
+ // Upload cap (when present)
308
+ if (m.max_file_size != null) {
309
+ const mb = Math.round(m.max_file_size / (1024 * 1024));
310
+ parts.push(`max_file_size: ${mb}MB`);
311
+ }
312
+
313
+ // Images-per-request (Midjourney-style fixed-N output)
314
+ if (m.images_per_request != null && m.images_per_request !== 1) {
315
+ parts.push(`images_per_request: ${m.images_per_request}`);
316
+ }
317
+
318
+ // Quality tiers (image models that support quality selection)
319
+ if (Array.isArray(m.supported_qualities) && m.supported_qualities.length) {
320
+ const qMult = m.quality_multipliers || {};
321
+ const qParts = m.supported_qualities.map(q =>
322
+ qMult[q] && qMult[q] !== 1 ? `${q}(${qMult[q]}×)` : q
323
+ );
324
+ parts.push(`quality: ${qParts.join(' · ')}${m.default_quality ? ` (default ${m.default_quality})` : ''}`);
325
+ }
326
+
327
+ // Fixed-price override (some models charge a flat rate per resolution instead of per-second)
328
+ if (m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object' && Object.keys(m.flat_credit_by_resolution).length) {
329
+ const fp = Object.entries(m.flat_credit_by_resolution).map(([k, v]) => `${k}:${v}cr`).join(' · ');
330
+ parts.push(`flat_price: ${fp}`);
331
+ }
332
+
333
+ // Estimated generation time (wall-clock at base settings)
334
+ if (m.estimated_duration_seconds != null) {
335
+ parts.push(`est_time: ~${m.estimated_duration_seconds}s`);
336
+ }
337
+
338
+ // NSFW flag
339
+ if (m.nsfw_only) {
340
+ parts.push('nsfw: required');
341
+ }
342
+
343
+ return parts.length ? `\n ${parts.join(' | ')}` : '';
344
+ };
345
+
346
+ // The FULL catalog with every spec line measured 140,590 chars — past what
347
+ // hosts accept, on the one discovery tool the skill tells the model to call
348
+ // when it is unsure. Unfiltered, emit the one-line form (enough to choose a
349
+ // model); once `type` narrows it, the set is small enough for full specs.
350
+ const detailed = !!type;
351
+ // Summaries run to a paragraph each; across the whole catalog that alone
352
+ // is most of the payload. Unfiltered, one clause is enough to choose by.
353
+ const brief = (s) => {
354
+ if (!s) return '';
355
+ const flat = String(s).replace(/\s+/g, ' ').trim();
356
+ return flat.length > 130 ? flat.slice(0, 127).trimEnd() + '…' : flat;
357
+ };
358
+ // Text models bill per token — the flat `credit` is not what the user pays,
359
+ // so show the real per-1K rates when the API supplies them. Without this the
360
+ // "cheapest model that fits" rule is unusable for chat.
361
+ //
362
+ // Video-type models are the same problem in a different shape: kolbo-api's
363
+ // credit engine (credManagment.js) treats "charge per second of requested
364
+ // duration" as the UNIVERSAL rule for any type in
365
+ // [video, firstlast, elements, motion_graphic, cast] — not a per-model
366
+ // exception, the default. So `credit: 9` on a model with duration 8 is
367
+ // really 72 credits, and nothing in the catalog said so: an agent quoting
368
+ // cost from the bare `credit` field alone is wrong by exactly the
369
+ // requested duration, every time. `flat_credit_by_resolution` is the one
370
+ // carve-out — those models are charged the flat rate regardless of
371
+ // duration, so they're excluded here the same way credManagment.js
372
+ // excludes them (resolveFlatCredit wins over the multiplier).
373
+ const PER_SECOND_TYPES = ['video', 'firstlast', 'elements', 'motion_graphic', 'cast'];
374
+ const isPerSecondVideo = m => {
375
+ const types = Array.isArray(m.types) ? m.types : (m.type ? [m.type] : []);
376
+ const billedPerSecond = types.some(t => PER_SECOND_TYPES.some(kw => String(t).includes(kw)));
377
+ const hasFlatOverride = m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object'
378
+ && Object.keys(m.flat_credit_by_resolution).length > 0;
379
+ return billedPerSecond && !hasFlatOverride;
380
+ };
381
+ const cost = m => (m.output_token_rate != null
382
+ ? `${m.input_token_rate ?? '?'}/${m.output_token_rate} credits per 1K tokens (in/out)`
383
+ : isPerSecondVideo(m)
384
+ ? `${m.credit} credits/output-second${m.video_input_credit != null ? `; ${m.video_input_credit} credits/combined-second with video input` : ''}`
385
+ : `${m.credit} credits`);
386
+ const formatModel = m =>
387
+ `${m.identifier} (${m.name}) - ${cost(m)}${m.recommended ? ' [RECOMMENDED]' : ''}${m.new_model ? ' [NEW]' : ''}${m.summary ? ` — ${detailed ? m.summary : brief(m.summary)}` : ''}${detailed ? formatSpecs(m) : ''}`;
388
+
389
+ const sections = [];
390
+
391
+ if (!detailed) {
392
+ // The catalog is ~428 models. Listing all of them is both far past the
393
+ // text budget AND useless to choose from — so unfiltered, surface the
394
+ // curated picks and make the model narrow by `type` for the rest. This
395
+ // matches the connector rule of steering to a CONCRETE model.
396
+ const picks = result.models.filter(m => m.recommended || m.new_model);
397
+ if (picks.length) {
398
+ sections.push(`Recommended & new (${picks.length}):\n${picks.map(formatModel).join('\n')}`);
399
+ }
400
+ const text = `Kolbo model catalog — ${result.count} models total.\n\n`
401
+ + `${sections.join('\n\n')}\n\n`
402
+ + 'This shortlist is BADGE-BASED (recommended/new) — it is not a recommendation to '
403
+ + 'use the newest or biggest model. To pick properly, re-call with `type` and choose by '
404
+ + 'each model\'s strengths summary, taking the cheapest one that covers the task. '
405
+ + 'To see everything in a '
406
+ + 'category (with per-model resolutions, durations, aspect ratios and reference-image '
407
+ + 'caps), re-call with `type`:\n'
408
+ + ' text_to_img · image_editing · text_to_video · img_to_video · video_to_video ·\n'
409
+ + ' first_last_frame · elements · lipsync · music_gen · text_to_speech ·\n'
410
+ + ' text_to_sound · stt · three_d · text\n\n'
411
+ + 'Use the "identifier" value as the "model" parameter in generate tools. '
412
+ + 'For EVERY model + its exact identifier, re-call with format: "json" (compact index of the '
413
+ + 'whole catalog). Add `type` to that call for the full raw documents with all caps.';
414
+ return respond(text);
415
+ }
416
+
417
+ if (withSummary.length > 0) {
418
+ sections.push(`Auto-selectable models (${withSummary.length}) — If the user already named a model/family (this turn or earlier), use that family — do not cheapest-swap. Otherwise CHOOSE BY THE SUMMARY after each "—": match it to what the user asked for, then take the CHEAPEST model that fits. Credit cost, [NEW] and [RECOMMENDED] are not reasons to pick a model:\n${withSummary.map(formatModel).join('\n')}`);
419
+ }
420
+ if (withoutSummary.length > 0) {
421
+ sections.push(`Named-only models (${withoutSummary.length}) — only use if the user explicitly requests by name:\n${withoutSummary.map(formatModel).join('\n')}`);
422
+ }
423
+
424
+ const text = `Available ${type} models (${result.count}):\n\n${sections.join('\n\n')}\n\nEvery ${type} model in the catalog is listed above — both sections together are the complete set. Use the "identifier" value as the "model" parameter in generate tools. For the raw documents (programmatic cap validation), re-call with format: "json".`;
425
+ return respond(text);
426
+ }
427
+ );
428
+
429
+ // ─── check_credits ─────────────────────────────────────────
430
+ server.tool(
431
+ 'check_credits',
432
+ 'Check your remaining Kolbo credit balance.',
433
+ {},
434
+ async () => {
435
+ const result = await client.get('/v1/account/credits');
436
+
437
+ return {
438
+ content: [{
439
+ type: 'text',
440
+ text: `Credit Balance:\n- Total: ${result.credits.total}\n- Plan credits: ${result.credits.plan_credits}\n- Credit pack: ${result.credits.credit_pack}\n- Redemption: ${result.credits.redemption}`
441
+ }]
442
+ };
443
+ }
444
+ );
445
+
446
+ // ─── get_session_usage ─────────────────────────────────────
447
+ // Real, multiplier-adjusted credit spend tagged with the caller's
448
+ // X-Kolbo-Caller-Session-Id (set automatically by the parent process —
449
+ // no need to pass it). Use this to give the user an honest "you've spent
450
+ // X credits in this app session" instead of estimating from base credits.
451
+ server.tool(
452
+ 'get_session_usage',
453
+ 'Fetch real, multiplier-adjusted credit spend for the current Kolbo Code app session. Use when the user asks "how much did I spend?" or before/after a large bulk job so you can quote actual cost (not an estimate from base credits). Returns total + per-tool breakdown + per-model breakdown + a recent list. The caller-session-id is forwarded automatically by the MCP HTTP client. ONLY works when running under Kolbo Code — on the claude.ai connector and other hosts there is no per-app session to scope to; use check_credits there instead.',
454
+ {},
455
+ async () => {
456
+ // Session scoping needs KOLBO_CALLER_SESSION_ID, which only the Kolbo Code
457
+ // parent process sets. The remote connector serves every user from one
458
+ // process, so it is never present there — calling anyway just returns a 400
459
+ // telling the user to reconfigure a process they do not control.
460
+ if (!process.env.KOLBO_CALLER_SESSION_ID) {
461
+ return {
462
+ content: [{
463
+ type: 'text',
464
+ text: JSON.stringify({
465
+ unavailable: 'Per-session usage is only tracked when running under Kolbo Code.',
466
+ reason: 'This host does not scope tool calls to an app session, so there is no session to total.',
467
+ use_instead: 'check_credits for the current balance, or the Usage page at https://app.kolbo.ai.'
468
+ })
469
+ }]
470
+ };
471
+ }
472
+ try {
473
+ const r = await client.get('/credit-usage/by-caller-session');
474
+ // The endpoint returns { message, data: { total, count, by_tool, by_model, recent[] } }
475
+ return {
476
+ content: [{
477
+ type: 'text',
478
+ text: JSON.stringify(r.data || r, null, 2)
479
+ }]
480
+ };
481
+ } catch (err) {
482
+ // 400 from the endpoint means no caller-session-id was forwarded —
483
+ // surface a clear hint instead of a generic API error.
484
+ const hint = err?.status === 400
485
+ ? 'No caller-session-id was forwarded. Ensure the parent process (Kolbo Code / desktop sidecar) sets KOLBO_CALLER_SESSION_ID in this MCP\'s env, or call again after at least one media generation has fired.'
486
+ : err?.message || 'Failed to fetch session usage';
487
+ return {
488
+ content: [{ type: 'text', text: JSON.stringify({ error: hint }, null, 2) }]
489
+ };
490
+ }
491
+ }
492
+ );
493
+
494
+ // ─── show_plans ──────────────────────────────────
495
+ // The upgrade card. Also rendered automatically when a generation is refused
496
+ // for credits — see insufficientCreditsResult() in _shared.js.
497
+ // Not registered at all under `commerce: false` (ChatGPT app profile): the
498
+ // OpenAI directory forbids selling digital goods, and this tool exists only
499
+ // to sell subscriptions and credit packs.
500
+ if (options.commerce === false) return;
501
+ server.tool(
502
+ 'show_plans',
503
+ 'Show the user their Kolbo credit balance, current plan, and the available upgrade plans / credit packs as an interactive card. Use when the user asks about pricing, plans, upgrading, or how to get more credits. Prices shown are live and promo-adjusted. The card links to app.kolbo.ai/pricing to complete a purchase — never quote prices from memory, and never claim to have made a purchase for them.',
504
+ {},
505
+ async () => {
506
+ const data = await client.get('/v1/account/plans');
507
+ const structured = {
508
+ widget: 'plans',
509
+ reason: 'requested',
510
+ balance: data?.credits?.total,
511
+ current_plan: data?.current_plan || null,
512
+ plans: data?.plans || [],
513
+ credit_packs: data?.credit_packs || [],
514
+ pricing_url: data?.pricing_url || 'https://app.kolbo.ai/pricing',
515
+ };
516
+ // structuredContent SHADOWS the text on widget hosts, so everything the
517
+ // agent needs to talk about pricing has to live in the object above.
518
+ const text = JSON.stringify({
519
+ credits: structured.balance,
520
+ current_plan: structured.current_plan,
521
+ plans: structured.plans,
522
+ credit_packs: structured.credit_packs,
523
+ pricing_url: structured.pricing_url,
524
+ _hint: 'A plans card is rendered for the user. Summarise briefly; do NOT paste the price table. Purchases are completed by the user on the pricing page — you cannot buy on their behalf.',
525
+ }, null, 2);
526
+ return uiResult(UI.plans, text, structured);
527
+ }
528
+ );
529
+ }
530
+
531
+ module.exports = { registerModelTools };