@kolbo/mcp 1.87.12 → 1.87.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skill/GENERATED.md +1 -1
- package/src/index.js +264 -257
- package/src/toolAnnotations.js +149 -145
- package/src/tools/_shared.js +1213 -1191
- package/src/tools/models.js +531 -527
package/src/tools/models.js
CHANGED
|
@@ -1,527 +1,531 @@
|
|
|
1
|
-
/* ⛔ BACKWARD COMPATIBILITY: Tool names and arg names below are a PUBLIC
|
|
2
|
-
* CONTRACT. Never rename, remove, or break an existing tool/arg — old cached
|
|
3
|
-
* `npx @kolbo/mcp` installs in the wild will break silently. Add new tools or
|
|
4
|
-
* new OPTIONAL args only. Full rules: ../index.js top-of-file and CLAUDE.md. */
|
|
5
|
-
|
|
6
|
-
const { z } = require('zod');
|
|
7
|
-
const { UI, uiResult, appsEnabled, resolveAvatarUrl } = require('../apps');
|
|
8
|
-
|
|
9
|
-
// type name → human group label for the catalog widget
|
|
10
|
-
const TYPE_GROUPS = {
|
|
11
|
-
text_to_img: 'Image Generation',
|
|
12
|
-
text_to_video: 'Video Generation',
|
|
13
|
-
img_to_video: 'Video Generation',
|
|
14
|
-
music_gen: 'Music',
|
|
15
|
-
text_to_speech: 'Voice',
|
|
16
|
-
image_editing: 'Image Editing',
|
|
17
|
-
video_to_video: 'Video to Video',
|
|
18
|
-
image_upscale: 'Image Upscale',
|
|
19
|
-
image_reframe: 'Image Reframe',
|
|
20
|
-
image_zoom_out: 'Image Expand',
|
|
21
|
-
video_upscale: 'Video Upscale',
|
|
22
|
-
video_reframe: 'Video Reframe',
|
|
23
|
-
video_background_removal: 'Video Background Removal',
|
|
24
|
-
video_to_sound: 'Video Audio Generation',
|
|
25
|
-
video_face_swap: 'Video Face Swap',
|
|
26
|
-
video_watermark_removal: 'Video Watermark Removal',
|
|
27
|
-
video_extend: 'Video Extend',
|
|
28
|
-
video_inpaint: 'Video Inpaint',
|
|
29
|
-
video_retake: 'Video Retake',
|
|
30
|
-
elements: 'Elements',
|
|
31
|
-
};
|
|
32
|
-
|
|
33
|
-
function groupNameFor(m) {
|
|
34
|
-
const t = (Array.isArray(m.types) && m.types[0]) || m.type || '';
|
|
35
|
-
if (TYPE_GROUPS[t]) return TYPE_GROUPS[t];
|
|
36
|
-
if (t === 'three_d' || String(t).startsWith('3d_')) return '3D';
|
|
37
|
-
return 'Other';
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
function modelChips(m) {
|
|
41
|
-
const chips = [];
|
|
42
|
-
if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
|
|
43
|
-
const highest = [...m.supported_resolutions]
|
|
44
|
-
.sort((a, b) => (parseInt(a, 10) || 0) - (parseInt(b, 10) || 0))
|
|
45
|
-
.pop();
|
|
46
|
-
if (highest) chips.push(String(highest));
|
|
47
|
-
}
|
|
48
|
-
if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
|
|
49
|
-
const ds = [...m.supported_durations].sort((a, b) => a - b);
|
|
50
|
-
chips.push(ds.length > 1 ? `${ds[0]}-${ds[ds.length - 1]}s` : `${ds[0]}s`);
|
|
51
|
-
}
|
|
52
|
-
if (m.supports_visual_dna) chips.push('DNA');
|
|
53
|
-
if (m.new_model || m.newModel) chips.push('NEW');
|
|
54
|
-
return chips.slice(0, 3);
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
// structuredContent for ui://kolbo/catalog.html — see src/apps/widgets/catalog.js
|
|
58
|
-
// Deliberately CURATED, not exhaustive: the widget is a picker, not a database.
|
|
59
|
-
// Each group shows the recommended/new models (max 6), and the total count
|
|
60
|
-
// chip tells the user how many exist overall. Smart Select / "Auto" rows are
|
|
61
|
-
// deliberately EXCLUDED — we always want a specific model chosen (server
|
|
62
|
-
// instruction #9): auto-routing hides the model choice and the generation
|
|
63
|
-
// metadata used to read just "Auto".
|
|
64
|
-
function buildCatalogStructured(models, type, compact) {
|
|
65
|
-
const groups = [];
|
|
66
|
-
const byName = new Map();
|
|
67
|
-
const isAuto = (m) => /^auto$|smart.select/i.test(String(m.name || '')) || /smart-select|k_auto/i.test(String(m.identifier || ''));
|
|
68
|
-
|
|
69
|
-
// Recommended + new models float to the top of each group.
|
|
70
|
-
const ranked = [...models].filter((m) => !isAuto(m)).sort((a, b) => {
|
|
71
|
-
const score = (m) => (m.recommended ? 2 : 0) + (m.new_model || m.newModel ? 1 : 0);
|
|
72
|
-
return score(b) - score(a);
|
|
73
|
-
});
|
|
74
|
-
|
|
75
|
-
for (const m of ranked) {
|
|
76
|
-
const name = groupNameFor(m);
|
|
77
|
-
let g = byName.get(name);
|
|
78
|
-
if (!g) { g = { name, models: [] }; byName.set(name, g); groups.push(g); }
|
|
79
|
-
if (g.models.length >= 6) continue; // curated cap — full list lives in the text payload
|
|
80
|
-
g.models.push({
|
|
81
|
-
name: m.name,
|
|
82
|
-
// The widget renders `name`; the AGENT reads the same rows (hosts hand it
|
|
83
|
-
// structuredContent). Without the identifier the default call was a dead
|
|
84
|
-
// end — it named six models and gave no way to pass any of them on.
|
|
85
|
-
identifier: m.identifier,
|
|
86
|
-
icon: resolveAvatarUrl(m.avatar),
|
|
87
|
-
description: String(m.smartSelect_StrengthsSummary || m.summary || m.description || '').slice(0, 90),
|
|
88
|
-
chips: modelChips(m),
|
|
89
|
-
use_hint: `Generate with the "${m.name}" model — ask me what I want to create first.`,
|
|
90
|
-
});
|
|
91
|
-
}
|
|
92
|
-
groups.sort((a, b) => (a.name === 'Other' ? 1 : b.name === 'Other' ? -1 : 0));
|
|
93
|
-
return {
|
|
94
|
-
widget: 'catalog',
|
|
95
|
-
title: 'Kolbo AI Models' + (type ? ' — ' + type : ''),
|
|
96
|
-
total_available: models.length,
|
|
97
|
-
compact: compact === true,
|
|
98
|
-
groups,
|
|
99
|
-
};
|
|
100
|
-
}
|
|
101
|
-
|
|
102
|
-
// One row per model — every identifier, nothing else. ~90 bytes/model, so the
|
|
103
|
-
// whole 400+ model catalog fits in a payload an agent can actually read.
|
|
104
|
-
const identifierRow = (m) => ({
|
|
105
|
-
identifier: m.identifier,
|
|
106
|
-
name: m.name,
|
|
107
|
-
types: m.types,
|
|
108
|
-
credit: m.credit,
|
|
109
|
-
...(m.video_input_credit != null ? { video_input_credit: m.video_input_credit } : {}),
|
|
110
|
-
...(m.recommended ? { recommended: true } : {}),
|
|
111
|
-
...(m.new_model ? { new_model: true } : {}),
|
|
112
|
-
});
|
|
113
|
-
|
|
114
|
-
function registerModelTools(server, client, options = {}) {
|
|
115
|
-
const ui = () => appsEnabled(server, options);
|
|
116
|
-
// ─── list_models ───────────────────────────────────────────
|
|
117
|
-
server.tool(
|
|
118
|
-
'list_models',
|
|
119
|
-
'List available AI models on Kolbo. Filter by `type` to narrow to a generation type, and pass `format: "json"` to enumerate the catalog with exact identifiers — `format: "json"` + `type` returns the full raw model documents (every constraint field, for programmatic comparison / cap validation before submitting a generation); `format: "json"` alone returns a compact index of EVERY model and its identifier. Default `format: "text"` returns the human-readable summary. NEVER guess a model identifier: call this tool. ⚠️ COST: video / firstlast / elements / motion_graphic / cast rates are normally per output second. If a model publishes `video_input_credit` and the request includes one or more input videos, use that alternate rate and bill nominal input seconds + nominal output seconds. Encoder padding within 0.15s of an integer snaps to it; larger fractions round up. A `flat_credit_by_resolution` model instead charges the flat tier regardless of duration. Every other model type (image, audio, 3D, per-token text) bills as its catalog fields state.',
|
|
120
|
-
{
|
|
121
|
-
type: z.string().optional().describe('Filter by DB type name. Generation: "text_to_img", "image_editing", "text_to_video", "img_to_video", "draw_to_video", "video_to_video", "elements", "firstlastgenerations", "lipsync-image", "lipsync-video", "music_gen", "text_to_speech", "text_to_sound", "stt", "text". Image-edit engines: "image_upscale", "image_reframe", "image_zoom_out", "inpaint", "erase", "face_swap", "background_remove", "background_replace", "skin_enhancer", "graphics_enhance". Video-edit engines: "video_upscale", "video_reframe", "video_background_removal", "video_to_sound", "video_face_swap", "video_watermark_removal", "video_extend", "video_inpaint", "video_retake". For edit_image/edit_video, query the operation-specific type and pass a CONCRETE returned identifier; never submit a kolbo_gateway_* row, because those are web-navigation aliases rather than AI engines. Legacy aliases also accepted: "image", "image_edit", "video", "video_from_image", "video_from_video", "music", "speech", "sound", "chat", "lipsync", "three_d", "first_last_frame", "transcription". Omit for all models.'),
|
|
122
|
-
format: z.enum(['text', 'json']).optional().describe('Output format. "text" (default) returns a human-readable summary with the most-used caps. "json" is the source of truth for identifiers and caps: with `type` it returns the raw model documents from the API (identifier, credit, supported_durations, supported_resolutions, supported_aspect_ratios, max_reference_images, max_visual_dna, max_video_duration, …) for EVERY model of that type; without `type` it returns a compact index of every model in the catalog and its exact identifier. Use it whenever you need an identifier you have not seen listed, or must verify a cap before passing a value that might exceed a model-specific limit.'),
|
|
123
|
-
display_catalog: z.boolean().optional().describe('Set true when the USER explicitly asked to see/browse the available models — the visual catalog opens expanded. Leave unset for internal lookups (verifying a model name, checking caps before a generation): the catalog stays collapsed to a single row the user can tap to browse.')
|
|
124
|
-
},
|
|
125
|
-
async ({ type, format, display_catalog }) => {
|
|
126
|
-
// The tool DECLARATION always carries widget meta, so hosts that mount
|
|
127
|
-
// from tools/list (Claude Code desktop) prepare an iframe on EVERY call.
|
|
128
|
-
// Returning plain text for internal lookups left that iframe with no
|
|
129
|
-
// data — a dead, empty "Widget from Kolbo list_models" shell. Always
|
|
130
|
-
// ship structuredContent; `compact` tells the widget to render a single
|
|
131
|
-
// "Browse models" row (expandable) instead of the full catalog, which is
|
|
132
|
-
// what display_catalog was really asking for.
|
|
133
|
-
const showCatalog = display_catalog === true;
|
|
134
|
-
const path = type ? `/v1/models?type=${encodeURIComponent(type)}` : '/v1/models';
|
|
135
|
-
const result = await client.get(path);
|
|
136
|
-
|
|
137
|
-
// ⚠️ Hosts that mount this widget (claude.ai, Claude Code desktop) hand the
|
|
138
|
-
// MODEL `structuredContent` and DROP `content[].text`. So every payload the
|
|
139
|
-
// agent needs has to ride in structuredContent — shipping it as text only
|
|
140
|
-
// makes it invisible. That is exactly how `format: "json"` came to return
|
|
141
|
-
// the curated 6-per-group picker instead of the raw documents: v1.53.1
|
|
142
|
-
// (406a51e) flipped `if (ui() && showCatalog)` → `if (ui())` on all three
|
|
143
|
-
// return paths, so the widget payload started shadowing the real answer and
|
|
144
|
-
// the other 43 text_to_video identifiers became undiscoverable by any MCP
|
|
145
|
-
// call. On 2026-08-09 that cost a wrong-model generation (minimax-h3).
|
|
146
|
-
// `extra` (json mode) carries the data as structured fields; without it the
|
|
147
|
-
// full text payload is attached verbatim. The widget ignores both.
|
|
148
|
-
// Always ship structuredContent, exactly like listResult() does and for
|
|
149
|
-
// the same reason: the tool DECLARATION carries widget meta, so a host
|
|
150
|
-
// mounts the catalog iframe on every call — including hosts that never
|
|
151
|
-
// advertised MCP Apps (Kolbo Code). Gating the payload on ui() left that
|
|
152
|
-
// iframe with nothing to render and it sat on "Loading..." forever.
|
|
153
|
-
const respond = (text, extra) =>
|
|
154
|
-
uiResult(UI.catalog, text, {
|
|
155
|
-
...buildCatalogStructured(result.models, type, !showCatalog),
|
|
156
|
-
...(extra || { text }),
|
|
157
|
-
});
|
|
158
|
-
|
|
159
|
-
// JSON mode — the authoritative shape; every constraint the agent might
|
|
160
|
-
// need to validate a request lives here (durations, reference caps,
|
|
161
|
-
// audio/video min/max, resolution multipliers, supports_* flags,
|
|
162
|
-
// prompt-length limits, etc.).
|
|
163
|
-
if (format === 'json') {
|
|
164
|
-
// Raw documents once `type` narrows the set (~49 docs for a video type).
|
|
165
|
-
// Unfiltered that is 400+ documents / hundreds of KB, so return the
|
|
166
|
-
// complete IDENTIFIER INDEX instead: every model stays enumerable and
|
|
167
|
-
// the full caps are one `type` away.
|
|
168
|
-
const payload = type
|
|
169
|
-
? { count: result.count, models: result.models }
|
|
170
|
-
: {
|
|
171
|
-
count: result.count,
|
|
172
|
-
models: result.models.map(identifierRow),
|
|
173
|
-
note: 'Compact index — every model in the catalog and its exact identifier. Re-call with `type` for the full documents (all caps, credit costs, supported_* fields).',
|
|
174
|
-
};
|
|
175
|
-
return respond(JSON.stringify(payload, null, 2), payload);
|
|
176
|
-
}
|
|
177
|
-
|
|
178
|
-
// Split into auto-selectable (has summary) and named-only (no summary)
|
|
179
|
-
const withSummary = result.models.filter(m => m.summary && m.summary.trim() !== '');
|
|
180
|
-
const withoutSummary = result.models.filter(m => !m.summary || m.summary.trim() === '');
|
|
181
|
-
|
|
182
|
-
// Format the per-model spec line. The agent NEEDS this — without it,
|
|
183
|
-
// it has to guess `supported_resolutions`/`supported_durations` and
|
|
184
|
-
// either invents values (then the API silently substitutes) or asks
|
|
185
|
-
// the user to clarify what's only knowable from this list.
|
|
186
|
-
//
|
|
187
|
-
// Rendering rule: emit a line for EVERY known constraint that is
|
|
188
|
-
// applicable for this model's type — even when the value is 0 / null.
|
|
189
|
-
// Hiding "0 cap" lines used to mean the agent couldn't distinguish
|
|
190
|
-
// "this model rejects DNA" (cap = 0) from "I don't know" (field
|
|
191
|
-
// missing). Now an explicit `max_dna: 0 (DNA not supported)` says the
|
|
192
|
-
// model says no, and absence means the API doesn't expose the field.
|
|
193
|
-
const formatSpecs = m => {
|
|
194
|
-
const parts = [];
|
|
195
|
-
if (m.haveThinking && Array.isArray(m.thinkingLevels) && m.thinkingLevels.length) {
|
|
196
|
-
parts.push(`thinking_level: ${m.thinkingLevels.map(level => level.id).join('/')} (default ${m.thinkingDefault})`);
|
|
197
|
-
}
|
|
198
|
-
const types = Array.isArray(m.types) ? m.types : [];
|
|
199
|
-
const isVideoType = types.some(t =>
|
|
200
|
-
['text_to_video', 'img_to_video', 'video_to_video', 'elements',
|
|
201
|
-
'firstlastgenerations', 'lipsync-image', 'lipsync-video', 'draw_to_video'].includes(t)
|
|
202
|
-
);
|
|
203
|
-
const isElements = types.includes('elements');
|
|
204
|
-
const isV2V = types.includes('video_to_video');
|
|
205
|
-
const isLipsyncVideo = types.includes('lipsync-video');
|
|
206
|
-
const isLipsyncImage = types.includes('lipsync-image');
|
|
207
|
-
const isImageEdit = types.includes('image_editing');
|
|
208
|
-
const isImage = types.includes('text_to_img') || isImageEdit;
|
|
209
|
-
|
|
210
|
-
if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
|
|
211
|
-
const mult = m.resolution_multipliers || {};
|
|
212
|
-
parts.push(
|
|
213
|
-
'resolutions: ' +
|
|
214
|
-
m.supported_resolutions
|
|
215
|
-
.map(r => (mult[r] != null && mult[r] !== 1 ? `${r} (${mult[r]}×)` : r))
|
|
216
|
-
.join(' · ')
|
|
217
|
-
);
|
|
218
|
-
}
|
|
219
|
-
|
|
220
|
-
if (m.video_input_credit != null) {
|
|
221
|
-
const vm = m.video_input_resolution_multipliers || {};
|
|
222
|
-
const tiers = Object.keys(vm).length
|
|
223
|
-
? ' · ' + Object.entries(vm).map(([r, mult]) => `${r} (${mult}×)`).join(' · ')
|
|
224
|
-
: '';
|
|
225
|
-
parts.push(`video_input_price: ${m.video_input_credit} credits/combined-second${tiers} · bill nominal input + output seconds (≤0.15s encoder padding snaps)`);
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
// Output durations (video gen output, not source video)
|
|
229
|
-
if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
|
|
230
|
-
const ds = m.supported_durations;
|
|
231
|
-
const sorted = [...ds].sort((a, b) => a - b);
|
|
232
|
-
const isRange = sorted.length > 2 && sorted.every((v, i) => i === 0 || v - sorted[i - 1] === 1);
|
|
233
|
-
parts.push(`durations: ${isRange ? `${sorted[0]}-${sorted[sorted.length - 1]}s` : sorted.join('/') + 's'}`);
|
|
234
|
-
} else if (isVideoType && (m.min_output_duration != null || m.max_output_duration != null)) {
|
|
235
|
-
parts.push(`duration_range: ${m.min_output_duration ?? '?'}-${m.max_output_duration ?? '?'}s${m.default_duration != null ? ` (default ${m.default_duration}s)` : ''}`);
|
|
236
|
-
}
|
|
237
|
-
|
|
238
|
-
// Aspect ratios — prefer per-type override if set
|
|
239
|
-
const ratios = m.supported_aspect_ratios_by_type
|
|
240
|
-
? Object.entries(m.supported_aspect_ratios_by_type).map(([t, arr]) => `${t}: ${arr.join('/')}`)
|
|
241
|
-
: null;
|
|
242
|
-
if (ratios) {
|
|
243
|
-
parts.push(`aspect (per-type): ${ratios.join(' | ')}`);
|
|
244
|
-
} else if (Array.isArray(m.supported_aspect_ratios) && m.supported_aspect_ratios.length) {
|
|
245
|
-
parts.push(`aspect: ${m.supported_aspect_ratios.join(', ')}${m.default_aspect_ratio ? ` (default ${m.default_aspect_ratio})` : ''}`);
|
|
246
|
-
}
|
|
247
|
-
|
|
248
|
-
// Reference-input caps — show the slot relevant for this model family.
|
|
249
|
-
// The same conceptual "max reference images" lives under THREE field
|
|
250
|
-
// names depending on the model type. Be explicit about which is which
|
|
251
|
-
// so the agent reads the right one.
|
|
252
|
-
if (isImage || isImageEdit) {
|
|
253
|
-
parts.push(`max_reference_images: ${m.max_reference_images ?? 0}${(m.max_reference_images ?? 0) === 0 ? ' (no refs)' : ''}`);
|
|
254
|
-
}
|
|
255
|
-
if (isElements) {
|
|
256
|
-
parts.push(`elements caps: imgs=${m.elements_max_images ?? 0} · vids=${m.elements_max_videos ?? 0} · audio=${m.elements_max_audio ?? 0}`);
|
|
257
|
-
}
|
|
258
|
-
if (isV2V) {
|
|
259
|
-
parts.push(`v2v ref caps: imgs=${m.max_images ?? 0} · vids=${m.max_videos ?? 0} · elements=${m.max_elements ?? 0} · audio=${m.max_audio ?? 0}`);
|
|
260
|
-
}
|
|
261
|
-
|
|
262
|
-
// Visual DNA cap — always show for image / elements / video, even if 0.
|
|
263
|
-
// Use the authoritative supports_visual_dna flag when available; fall
|
|
264
|
-
// back to inferring from cap > 0 for older API responses.
|
|
265
|
-
const dnaSupported = typeof m.supports_visual_dna === 'boolean'
|
|
266
|
-
? m.supports_visual_dna
|
|
267
|
-
: (m.max_visual_dna ?? 0) > 0;
|
|
268
|
-
if (isImage || isVideoType) {
|
|
269
|
-
const cap = m.max_visual_dna;
|
|
270
|
-
if (dnaSupported && cap != null && cap > 0) parts.push(`max_visual_dna: ${cap}`);
|
|
271
|
-
else if (dnaSupported && cap == null) parts.push('visual_dna: supported (no cap published — confirm before passing >3)');
|
|
272
|
-
else parts.push('visual_dna: not supported');
|
|
273
|
-
}
|
|
274
|
-
|
|
275
|
-
// Source-video duration constraints — only matter for tools that take
|
|
276
|
-
// an INPUT video (lipsync-video, video_to_video).
|
|
277
|
-
if (isLipsyncVideo || isV2V) {
|
|
278
|
-
if (m.min_video_duration != null || m.max_video_duration != null) {
|
|
279
|
-
parts.push(`source_video: ${m.min_video_duration ?? '?'}-${m.max_video_duration ?? '?'}s`);
|
|
280
|
-
}
|
|
281
|
-
}
|
|
282
|
-
|
|
283
|
-
// Audio input — lipsync, elements, music-driven flows.
|
|
284
|
-
if (m.max_audio_duration != null || m.min_audio_duration != null) {
|
|
285
|
-
parts.push(`audio_input: ${m.min_audio_duration ?? '?'}-${m.max_audio_duration ?? '?'}s${m.audio_max_follows_video_duration ? ' (max follows video)' : ''}`);
|
|
286
|
-
}
|
|
287
|
-
if (Array.isArray(m.supported_audio_formats) && m.supported_audio_formats.length) {
|
|
288
|
-
parts.push(`audio_formats: ${m.supported_audio_formats.join('/')}`);
|
|
289
|
-
}
|
|
290
|
-
|
|
291
|
-
// Native sound generation (video models that emit synced audio)
|
|
292
|
-
if (m.sound_generation_type === 'native') {
|
|
293
|
-
const mult = m.sound_credit_multiplier && m.sound_credit_multiplier !== 1
|
|
294
|
-
? ` (${m.sound_credit_multiplier}×)`
|
|
295
|
-
: '';
|
|
296
|
-
parts.push(`sound: native${mult}${m.sound_enabled_by_default ? ' on-by-default' : ''}`);
|
|
297
|
-
} else if (m.sound_baked_in) {
|
|
298
|
-
parts.push('sound: baked-in (always on; type=none hides the toggle — still real audio)');
|
|
299
|
-
}
|
|
300
|
-
|
|
301
|
-
// Prompt constraints
|
|
302
|
-
if (m.requires_prompt === false) parts.push('prompt: optional');
|
|
303
|
-
if (m.min_prompt_length != null || m.max_prompt_length != null) {
|
|
304
|
-
parts.push(`prompt_length: ${m.min_prompt_length ?? 0}-${m.max_prompt_length ?? '∞'} chars`);
|
|
305
|
-
}
|
|
306
|
-
|
|
307
|
-
// Upload cap (when present)
|
|
308
|
-
if (m.max_file_size != null) {
|
|
309
|
-
const mb = Math.round(m.max_file_size / (1024 * 1024));
|
|
310
|
-
parts.push(`max_file_size: ${mb}MB`);
|
|
311
|
-
}
|
|
312
|
-
|
|
313
|
-
// Images-per-request (Midjourney-style fixed-N output)
|
|
314
|
-
if (m.images_per_request != null && m.images_per_request !== 1) {
|
|
315
|
-
parts.push(`images_per_request: ${m.images_per_request}`);
|
|
316
|
-
}
|
|
317
|
-
|
|
318
|
-
// Quality tiers (image models that support quality selection)
|
|
319
|
-
if (Array.isArray(m.supported_qualities) && m.supported_qualities.length) {
|
|
320
|
-
const qMult = m.quality_multipliers || {};
|
|
321
|
-
const qParts = m.supported_qualities.map(q =>
|
|
322
|
-
qMult[q] && qMult[q] !== 1 ? `${q}(${qMult[q]}×)` : q
|
|
323
|
-
);
|
|
324
|
-
parts.push(`quality: ${qParts.join(' · ')}${m.default_quality ? ` (default ${m.default_quality})` : ''}`);
|
|
325
|
-
}
|
|
326
|
-
|
|
327
|
-
// Fixed-price override (some models charge a flat rate per resolution instead of per-second)
|
|
328
|
-
if (m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object' && Object.keys(m.flat_credit_by_resolution).length) {
|
|
329
|
-
const fp = Object.entries(m.flat_credit_by_resolution).map(([k, v]) => `${k}:${v}cr`).join(' · ');
|
|
330
|
-
parts.push(`flat_price: ${fp}`);
|
|
331
|
-
}
|
|
332
|
-
|
|
333
|
-
// Estimated generation time (wall-clock at base settings)
|
|
334
|
-
if (m.estimated_duration_seconds != null) {
|
|
335
|
-
parts.push(`est_time: ~${m.estimated_duration_seconds}s`);
|
|
336
|
-
}
|
|
337
|
-
|
|
338
|
-
// NSFW flag
|
|
339
|
-
if (m.nsfw_only) {
|
|
340
|
-
parts.push('nsfw: required');
|
|
341
|
-
}
|
|
342
|
-
|
|
343
|
-
return parts.length ? `\n ${parts.join(' | ')}` : '';
|
|
344
|
-
};
|
|
345
|
-
|
|
346
|
-
// The FULL catalog with every spec line measured 140,590 chars — past what
|
|
347
|
-
// hosts accept, on the one discovery tool the skill tells the model to call
|
|
348
|
-
// when it is unsure. Unfiltered, emit the one-line form (enough to choose a
|
|
349
|
-
// model); once `type` narrows it, the set is small enough for full specs.
|
|
350
|
-
const detailed = !!type;
|
|
351
|
-
// Summaries run to a paragraph each; across the whole catalog that alone
|
|
352
|
-
// is most of the payload. Unfiltered, one clause is enough to choose by.
|
|
353
|
-
const brief = (s) => {
|
|
354
|
-
if (!s) return '';
|
|
355
|
-
const flat = String(s).replace(/\s+/g, ' ').trim();
|
|
356
|
-
return flat.length > 130 ? flat.slice(0, 127).trimEnd() + '…' : flat;
|
|
357
|
-
};
|
|
358
|
-
// Text models bill per token — the flat `credit` is not what the user pays,
|
|
359
|
-
// so show the real per-1K rates when the API supplies them. Without this the
|
|
360
|
-
// "cheapest model that fits" rule is unusable for chat.
|
|
361
|
-
//
|
|
362
|
-
// Video-type models are the same problem in a different shape: kolbo-api's
|
|
363
|
-
// credit engine (credManagment.js) treats "charge per second of requested
|
|
364
|
-
// duration" as the UNIVERSAL rule for any type in
|
|
365
|
-
// [video, firstlast, elements, motion_graphic, cast] — not a per-model
|
|
366
|
-
// exception, the default. So `credit: 9` on a model with duration 8 is
|
|
367
|
-
// really 72 credits, and nothing in the catalog said so: an agent quoting
|
|
368
|
-
// cost from the bare `credit` field alone is wrong by exactly the
|
|
369
|
-
// requested duration, every time. `flat_credit_by_resolution` is the one
|
|
370
|
-
// carve-out — those models are charged the flat rate regardless of
|
|
371
|
-
// duration, so they're excluded here the same way credManagment.js
|
|
372
|
-
// excludes them (resolveFlatCredit wins over the multiplier).
|
|
373
|
-
const PER_SECOND_TYPES = ['video', 'firstlast', 'elements', 'motion_graphic', 'cast'];
|
|
374
|
-
const isPerSecondVideo = m => {
|
|
375
|
-
const types = Array.isArray(m.types) ? m.types : (m.type ? [m.type] : []);
|
|
376
|
-
const billedPerSecond = types.some(t => PER_SECOND_TYPES.some(kw => String(t).includes(kw)));
|
|
377
|
-
const hasFlatOverride = m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object'
|
|
378
|
-
&& Object.keys(m.flat_credit_by_resolution).length > 0;
|
|
379
|
-
return billedPerSecond && !hasFlatOverride;
|
|
380
|
-
};
|
|
381
|
-
const cost = m => (m.output_token_rate != null
|
|
382
|
-
? `${m.input_token_rate ?? '?'}/${m.output_token_rate} credits per 1K tokens (in/out)`
|
|
383
|
-
: isPerSecondVideo(m)
|
|
384
|
-
? `${m.credit} credits/output-second${m.video_input_credit != null ? `; ${m.video_input_credit} credits/combined-second with video input` : ''}`
|
|
385
|
-
: `${m.credit} credits`);
|
|
386
|
-
const formatModel = m =>
|
|
387
|
-
`${m.identifier} (${m.name}) - ${cost(m)}${m.recommended ? ' [RECOMMENDED]' : ''}${m.new_model ? ' [NEW]' : ''}${m.summary ? ` — ${detailed ? m.summary : brief(m.summary)}` : ''}${detailed ? formatSpecs(m) : ''}`;
|
|
388
|
-
|
|
389
|
-
const sections = [];
|
|
390
|
-
|
|
391
|
-
if (!detailed) {
|
|
392
|
-
// The catalog is ~428 models. Listing all of them is both far past the
|
|
393
|
-
// text budget AND useless to choose from — so unfiltered, surface the
|
|
394
|
-
// curated picks and make the model narrow by `type` for the rest. This
|
|
395
|
-
// matches the connector rule of steering to a CONCRETE model.
|
|
396
|
-
const picks = result.models.filter(m => m.recommended || m.new_model);
|
|
397
|
-
if (picks.length) {
|
|
398
|
-
sections.push(`Recommended & new (${picks.length}):\n${picks.map(formatModel).join('\n')}`);
|
|
399
|
-
}
|
|
400
|
-
const text = `Kolbo model catalog — ${result.count} models total.\n\n`
|
|
401
|
-
+ `${sections.join('\n\n')}\n\n`
|
|
402
|
-
+ 'This shortlist is BADGE-BASED (recommended/new) — it is not a recommendation to '
|
|
403
|
-
+ 'use the newest or biggest model. To pick properly, re-call with `type` and choose by '
|
|
404
|
-
+ 'each model\'s strengths summary, taking the cheapest one that covers the task. '
|
|
405
|
-
+ 'To see everything in a '
|
|
406
|
-
+ 'category (with per-model resolutions, durations, aspect ratios and reference-image '
|
|
407
|
-
+ 'caps), re-call with `type`:\n'
|
|
408
|
-
+ ' text_to_img · image_editing · text_to_video · img_to_video · video_to_video ·\n'
|
|
409
|
-
+ ' first_last_frame · elements · lipsync · music_gen · text_to_speech ·\n'
|
|
410
|
-
+ ' text_to_sound · stt · three_d · text\n\n'
|
|
411
|
-
+ 'Use the "identifier" value as the "model" parameter in generate tools. '
|
|
412
|
-
+ 'For EVERY model + its exact identifier, re-call with format: "json" (compact index of the '
|
|
413
|
-
+ 'whole catalog). Add `type` to that call for the full raw documents with all caps.';
|
|
414
|
-
return respond(text);
|
|
415
|
-
}
|
|
416
|
-
|
|
417
|
-
if (withSummary.length > 0) {
|
|
418
|
-
sections.push(`Auto-selectable models (${withSummary.length}) — If the user already named a model/family (this turn or earlier), use that family — do not cheapest-swap. Otherwise CHOOSE BY THE SUMMARY after each "—": match it to what the user asked for, then take the CHEAPEST model that fits. Credit cost, [NEW] and [RECOMMENDED] are not reasons to pick a model:\n${withSummary.map(formatModel).join('\n')}`);
|
|
419
|
-
}
|
|
420
|
-
if (withoutSummary.length > 0) {
|
|
421
|
-
sections.push(`Named-only models (${withoutSummary.length}) — only use if the user explicitly requests by name:\n${withoutSummary.map(formatModel).join('\n')}`);
|
|
422
|
-
}
|
|
423
|
-
|
|
424
|
-
const text = `Available ${type} models (${result.count}):\n\n${sections.join('\n\n')}\n\nEvery ${type} model in the catalog is listed above — both sections together are the complete set. Use the "identifier" value as the "model" parameter in generate tools. For the raw documents (programmatic cap validation), re-call with format: "json".`;
|
|
425
|
-
return respond(text);
|
|
426
|
-
}
|
|
427
|
-
);
|
|
428
|
-
|
|
429
|
-
// ─── check_credits ─────────────────────────────────────────
|
|
430
|
-
server.tool(
|
|
431
|
-
'check_credits',
|
|
432
|
-
'Check your remaining Kolbo credit balance.',
|
|
433
|
-
{},
|
|
434
|
-
async () => {
|
|
435
|
-
const result = await client.get('/v1/account/credits');
|
|
436
|
-
|
|
437
|
-
return {
|
|
438
|
-
content: [{
|
|
439
|
-
type: 'text',
|
|
440
|
-
text: `Credit Balance:\n- Total: ${result.credits.total}\n- Plan credits: ${result.credits.plan_credits}\n- Credit pack: ${result.credits.credit_pack}\n- Redemption: ${result.credits.redemption}`
|
|
441
|
-
}]
|
|
442
|
-
};
|
|
443
|
-
}
|
|
444
|
-
);
|
|
445
|
-
|
|
446
|
-
// ─── get_session_usage ─────────────────────────────────────
|
|
447
|
-
// Real, multiplier-adjusted credit spend tagged with the caller's
|
|
448
|
-
// X-Kolbo-Caller-Session-Id (set automatically by the parent process —
|
|
449
|
-
// no need to pass it). Use this to give the user an honest "you've spent
|
|
450
|
-
// X credits in this app session" instead of estimating from base credits.
|
|
451
|
-
server.tool(
|
|
452
|
-
'get_session_usage',
|
|
453
|
-
'Fetch real, multiplier-adjusted credit spend for the current Kolbo Code app session. Use when the user asks "how much did I spend?" or before/after a large bulk job so you can quote actual cost (not an estimate from base credits). Returns total + per-tool breakdown + per-model breakdown + a recent list. The caller-session-id is forwarded automatically by the MCP HTTP client. ONLY works when running under Kolbo Code — on the claude.ai connector and other hosts there is no per-app session to scope to; use check_credits there instead.',
|
|
454
|
-
{},
|
|
455
|
-
async () => {
|
|
456
|
-
// Session scoping needs KOLBO_CALLER_SESSION_ID, which only the Kolbo Code
|
|
457
|
-
// parent process sets. The remote connector serves every user from one
|
|
458
|
-
// process, so it is never present there — calling anyway just returns a 400
|
|
459
|
-
// telling the user to reconfigure a process they do not control.
|
|
460
|
-
if (!process.env.KOLBO_CALLER_SESSION_ID) {
|
|
461
|
-
return {
|
|
462
|
-
content: [{
|
|
463
|
-
type: 'text',
|
|
464
|
-
text: JSON.stringify({
|
|
465
|
-
unavailable: 'Per-session usage is only tracked when running under Kolbo Code.',
|
|
466
|
-
reason: 'This host does not scope tool calls to an app session, so there is no session to total.',
|
|
467
|
-
use_instead: 'check_credits for the current balance, or the Usage page at https://app.kolbo.ai.'
|
|
468
|
-
})
|
|
469
|
-
}]
|
|
470
|
-
};
|
|
471
|
-
}
|
|
472
|
-
try {
|
|
473
|
-
const r = await client.get('/credit-usage/by-caller-session');
|
|
474
|
-
// The endpoint returns { message, data: { total, count, by_tool, by_model, recent[] } }
|
|
475
|
-
return {
|
|
476
|
-
content: [{
|
|
477
|
-
type: 'text',
|
|
478
|
-
text: JSON.stringify(r.data || r, null, 2)
|
|
479
|
-
}]
|
|
480
|
-
};
|
|
481
|
-
} catch (err) {
|
|
482
|
-
// 400 from the endpoint means no caller-session-id was forwarded —
|
|
483
|
-
// surface a clear hint instead of a generic API error.
|
|
484
|
-
const hint = err?.status === 400
|
|
485
|
-
? 'No caller-session-id was forwarded. Ensure the parent process (Kolbo Code / desktop sidecar) sets KOLBO_CALLER_SESSION_ID in this MCP\'s env, or call again after at least one media generation has fired.'
|
|
486
|
-
: err?.message || 'Failed to fetch session usage';
|
|
487
|
-
return {
|
|
488
|
-
content: [{ type: 'text', text: JSON.stringify({ error: hint }, null, 2) }]
|
|
489
|
-
};
|
|
490
|
-
}
|
|
491
|
-
}
|
|
492
|
-
);
|
|
493
|
-
|
|
494
|
-
// ─── show_plans ──────────────────────────────────
|
|
495
|
-
// The upgrade card. Also rendered automatically when a generation is refused
|
|
496
|
-
// for credits — see insufficientCreditsResult() in _shared.js.
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
}
|
|
526
|
-
|
|
527
|
-
|
|
1
|
+
/* ⛔ BACKWARD COMPATIBILITY: Tool names and arg names below are a PUBLIC
|
|
2
|
+
* CONTRACT. Never rename, remove, or break an existing tool/arg — old cached
|
|
3
|
+
* `npx @kolbo/mcp` installs in the wild will break silently. Add new tools or
|
|
4
|
+
* new OPTIONAL args only. Full rules: ../index.js top-of-file and CLAUDE.md. */
|
|
5
|
+
|
|
6
|
+
const { z } = require('zod');
|
|
7
|
+
const { UI, uiResult, appsEnabled, resolveAvatarUrl } = require('../apps');
|
|
8
|
+
|
|
9
|
+
// type name → human group label for the catalog widget
|
|
10
|
+
const TYPE_GROUPS = {
|
|
11
|
+
text_to_img: 'Image Generation',
|
|
12
|
+
text_to_video: 'Video Generation',
|
|
13
|
+
img_to_video: 'Video Generation',
|
|
14
|
+
music_gen: 'Music',
|
|
15
|
+
text_to_speech: 'Voice',
|
|
16
|
+
image_editing: 'Image Editing',
|
|
17
|
+
video_to_video: 'Video to Video',
|
|
18
|
+
image_upscale: 'Image Upscale',
|
|
19
|
+
image_reframe: 'Image Reframe',
|
|
20
|
+
image_zoom_out: 'Image Expand',
|
|
21
|
+
video_upscale: 'Video Upscale',
|
|
22
|
+
video_reframe: 'Video Reframe',
|
|
23
|
+
video_background_removal: 'Video Background Removal',
|
|
24
|
+
video_to_sound: 'Video Audio Generation',
|
|
25
|
+
video_face_swap: 'Video Face Swap',
|
|
26
|
+
video_watermark_removal: 'Video Watermark Removal',
|
|
27
|
+
video_extend: 'Video Extend',
|
|
28
|
+
video_inpaint: 'Video Inpaint',
|
|
29
|
+
video_retake: 'Video Retake',
|
|
30
|
+
elements: 'Elements',
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
function groupNameFor(m) {
|
|
34
|
+
const t = (Array.isArray(m.types) && m.types[0]) || m.type || '';
|
|
35
|
+
if (TYPE_GROUPS[t]) return TYPE_GROUPS[t];
|
|
36
|
+
if (t === 'three_d' || String(t).startsWith('3d_')) return '3D';
|
|
37
|
+
return 'Other';
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function modelChips(m) {
|
|
41
|
+
const chips = [];
|
|
42
|
+
if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
|
|
43
|
+
const highest = [...m.supported_resolutions]
|
|
44
|
+
.sort((a, b) => (parseInt(a, 10) || 0) - (parseInt(b, 10) || 0))
|
|
45
|
+
.pop();
|
|
46
|
+
if (highest) chips.push(String(highest));
|
|
47
|
+
}
|
|
48
|
+
if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
|
|
49
|
+
const ds = [...m.supported_durations].sort((a, b) => a - b);
|
|
50
|
+
chips.push(ds.length > 1 ? `${ds[0]}-${ds[ds.length - 1]}s` : `${ds[0]}s`);
|
|
51
|
+
}
|
|
52
|
+
if (m.supports_visual_dna) chips.push('DNA');
|
|
53
|
+
if (m.new_model || m.newModel) chips.push('NEW');
|
|
54
|
+
return chips.slice(0, 3);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
// structuredContent for ui://kolbo/catalog.html — see src/apps/widgets/catalog.js
|
|
58
|
+
// Deliberately CURATED, not exhaustive: the widget is a picker, not a database.
|
|
59
|
+
// Each group shows the recommended/new models (max 6), and the total count
|
|
60
|
+
// chip tells the user how many exist overall. Smart Select / "Auto" rows are
|
|
61
|
+
// deliberately EXCLUDED — we always want a specific model chosen (server
|
|
62
|
+
// instruction #9): auto-routing hides the model choice and the generation
|
|
63
|
+
// metadata used to read just "Auto".
|
|
64
|
+
function buildCatalogStructured(models, type, compact) {
|
|
65
|
+
const groups = [];
|
|
66
|
+
const byName = new Map();
|
|
67
|
+
const isAuto = (m) => /^auto$|smart.select/i.test(String(m.name || '')) || /smart-select|k_auto/i.test(String(m.identifier || ''));
|
|
68
|
+
|
|
69
|
+
// Recommended + new models float to the top of each group.
|
|
70
|
+
const ranked = [...models].filter((m) => !isAuto(m)).sort((a, b) => {
|
|
71
|
+
const score = (m) => (m.recommended ? 2 : 0) + (m.new_model || m.newModel ? 1 : 0);
|
|
72
|
+
return score(b) - score(a);
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
for (const m of ranked) {
|
|
76
|
+
const name = groupNameFor(m);
|
|
77
|
+
let g = byName.get(name);
|
|
78
|
+
if (!g) { g = { name, models: [] }; byName.set(name, g); groups.push(g); }
|
|
79
|
+
if (g.models.length >= 6) continue; // curated cap — full list lives in the text payload
|
|
80
|
+
g.models.push({
|
|
81
|
+
name: m.name,
|
|
82
|
+
// The widget renders `name`; the AGENT reads the same rows (hosts hand it
|
|
83
|
+
// structuredContent). Without the identifier the default call was a dead
|
|
84
|
+
// end — it named six models and gave no way to pass any of them on.
|
|
85
|
+
identifier: m.identifier,
|
|
86
|
+
icon: resolveAvatarUrl(m.avatar),
|
|
87
|
+
description: String(m.smartSelect_StrengthsSummary || m.summary || m.description || '').slice(0, 90),
|
|
88
|
+
chips: modelChips(m),
|
|
89
|
+
use_hint: `Generate with the "${m.name}" model — ask me what I want to create first.`,
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
groups.sort((a, b) => (a.name === 'Other' ? 1 : b.name === 'Other' ? -1 : 0));
|
|
93
|
+
return {
|
|
94
|
+
widget: 'catalog',
|
|
95
|
+
title: 'Kolbo AI Models' + (type ? ' — ' + type : ''),
|
|
96
|
+
total_available: models.length,
|
|
97
|
+
compact: compact === true,
|
|
98
|
+
groups,
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
// One row per model — every identifier, nothing else. ~90 bytes/model, so the
|
|
103
|
+
// whole 400+ model catalog fits in a payload an agent can actually read.
|
|
104
|
+
const identifierRow = (m) => ({
|
|
105
|
+
identifier: m.identifier,
|
|
106
|
+
name: m.name,
|
|
107
|
+
types: m.types,
|
|
108
|
+
credit: m.credit,
|
|
109
|
+
...(m.video_input_credit != null ? { video_input_credit: m.video_input_credit } : {}),
|
|
110
|
+
...(m.recommended ? { recommended: true } : {}),
|
|
111
|
+
...(m.new_model ? { new_model: true } : {}),
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
function registerModelTools(server, client, options = {}) {
|
|
115
|
+
const ui = () => appsEnabled(server, options);
|
|
116
|
+
// ─── list_models ───────────────────────────────────────────
|
|
117
|
+
server.tool(
|
|
118
|
+
'list_models',
|
|
119
|
+
'List available AI models on Kolbo. Filter by `type` to narrow to a generation type, and pass `format: "json"` to enumerate the catalog with exact identifiers — `format: "json"` + `type` returns the full raw model documents (every constraint field, for programmatic comparison / cap validation before submitting a generation); `format: "json"` alone returns a compact index of EVERY model and its identifier. Default `format: "text"` returns the human-readable summary. NEVER guess a model identifier: call this tool. ⚠️ COST: video / firstlast / elements / motion_graphic / cast rates are normally per output second. If a model publishes `video_input_credit` and the request includes one or more input videos, use that alternate rate and bill nominal input seconds + nominal output seconds. Encoder padding within 0.15s of an integer snaps to it; larger fractions round up. A `flat_credit_by_resolution` model instead charges the flat tier regardless of duration. Every other model type (image, audio, 3D, per-token text) bills as its catalog fields state.',
|
|
120
|
+
{
|
|
121
|
+
type: z.string().optional().describe('Filter by DB type name. Generation: "text_to_img", "image_editing", "text_to_video", "img_to_video", "draw_to_video", "video_to_video", "elements", "firstlastgenerations", "lipsync-image", "lipsync-video", "music_gen", "text_to_speech", "text_to_sound", "stt", "text". Image-edit engines: "image_upscale", "image_reframe", "image_zoom_out", "inpaint", "erase", "face_swap", "background_remove", "background_replace", "skin_enhancer", "graphics_enhance". Video-edit engines: "video_upscale", "video_reframe", "video_background_removal", "video_to_sound", "video_face_swap", "video_watermark_removal", "video_extend", "video_inpaint", "video_retake". For edit_image/edit_video, query the operation-specific type and pass a CONCRETE returned identifier; never submit a kolbo_gateway_* row, because those are web-navigation aliases rather than AI engines. Legacy aliases also accepted: "image", "image_edit", "video", "video_from_image", "video_from_video", "music", "speech", "sound", "chat", "lipsync", "three_d", "first_last_frame", "transcription". Omit for all models.'),
|
|
122
|
+
format: z.enum(['text', 'json']).optional().describe('Output format. "text" (default) returns a human-readable summary with the most-used caps. "json" is the source of truth for identifiers and caps: with `type` it returns the raw model documents from the API (identifier, credit, supported_durations, supported_resolutions, supported_aspect_ratios, max_reference_images, max_visual_dna, max_video_duration, …) for EVERY model of that type; without `type` it returns a compact index of every model in the catalog and its exact identifier. Use it whenever you need an identifier you have not seen listed, or must verify a cap before passing a value that might exceed a model-specific limit.'),
|
|
123
|
+
display_catalog: z.boolean().optional().describe('Set true when the USER explicitly asked to see/browse the available models — the visual catalog opens expanded. Leave unset for internal lookups (verifying a model name, checking caps before a generation): the catalog stays collapsed to a single row the user can tap to browse.')
|
|
124
|
+
},
|
|
125
|
+
async ({ type, format, display_catalog }) => {
|
|
126
|
+
// The tool DECLARATION always carries widget meta, so hosts that mount
|
|
127
|
+
// from tools/list (Claude Code desktop) prepare an iframe on EVERY call.
|
|
128
|
+
// Returning plain text for internal lookups left that iframe with no
|
|
129
|
+
// data — a dead, empty "Widget from Kolbo list_models" shell. Always
|
|
130
|
+
// ship structuredContent; `compact` tells the widget to render a single
|
|
131
|
+
// "Browse models" row (expandable) instead of the full catalog, which is
|
|
132
|
+
// what display_catalog was really asking for.
|
|
133
|
+
const showCatalog = display_catalog === true;
|
|
134
|
+
const path = type ? `/v1/models?type=${encodeURIComponent(type)}` : '/v1/models';
|
|
135
|
+
const result = await client.get(path);
|
|
136
|
+
|
|
137
|
+
// ⚠️ Hosts that mount this widget (claude.ai, Claude Code desktop) hand the
|
|
138
|
+
// MODEL `structuredContent` and DROP `content[].text`. So every payload the
|
|
139
|
+
// agent needs has to ride in structuredContent — shipping it as text only
|
|
140
|
+
// makes it invisible. That is exactly how `format: "json"` came to return
|
|
141
|
+
// the curated 6-per-group picker instead of the raw documents: v1.53.1
|
|
142
|
+
// (406a51e) flipped `if (ui() && showCatalog)` → `if (ui())` on all three
|
|
143
|
+
// return paths, so the widget payload started shadowing the real answer and
|
|
144
|
+
// the other 43 text_to_video identifiers became undiscoverable by any MCP
|
|
145
|
+
// call. On 2026-08-09 that cost a wrong-model generation (minimax-h3).
|
|
146
|
+
// `extra` (json mode) carries the data as structured fields; without it the
|
|
147
|
+
// full text payload is attached verbatim. The widget ignores both.
|
|
148
|
+
// Always ship structuredContent, exactly like listResult() does and for
|
|
149
|
+
// the same reason: the tool DECLARATION carries widget meta, so a host
|
|
150
|
+
// mounts the catalog iframe on every call — including hosts that never
|
|
151
|
+
// advertised MCP Apps (Kolbo Code). Gating the payload on ui() left that
|
|
152
|
+
// iframe with nothing to render and it sat on "Loading..." forever.
|
|
153
|
+
const respond = (text, extra) =>
|
|
154
|
+
uiResult(UI.catalog, text, {
|
|
155
|
+
...buildCatalogStructured(result.models, type, !showCatalog),
|
|
156
|
+
...(extra || { text }),
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
// JSON mode — the authoritative shape; every constraint the agent might
|
|
160
|
+
// need to validate a request lives here (durations, reference caps,
|
|
161
|
+
// audio/video min/max, resolution multipliers, supports_* flags,
|
|
162
|
+
// prompt-length limits, etc.).
|
|
163
|
+
if (format === 'json') {
|
|
164
|
+
// Raw documents once `type` narrows the set (~49 docs for a video type).
|
|
165
|
+
// Unfiltered that is 400+ documents / hundreds of KB, so return the
|
|
166
|
+
// complete IDENTIFIER INDEX instead: every model stays enumerable and
|
|
167
|
+
// the full caps are one `type` away.
|
|
168
|
+
const payload = type
|
|
169
|
+
? { count: result.count, models: result.models }
|
|
170
|
+
: {
|
|
171
|
+
count: result.count,
|
|
172
|
+
models: result.models.map(identifierRow),
|
|
173
|
+
note: 'Compact index — every model in the catalog and its exact identifier. Re-call with `type` for the full documents (all caps, credit costs, supported_* fields).',
|
|
174
|
+
};
|
|
175
|
+
return respond(JSON.stringify(payload, null, 2), payload);
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// Split into auto-selectable (has summary) and named-only (no summary)
|
|
179
|
+
const withSummary = result.models.filter(m => m.summary && m.summary.trim() !== '');
|
|
180
|
+
const withoutSummary = result.models.filter(m => !m.summary || m.summary.trim() === '');
|
|
181
|
+
|
|
182
|
+
// Format the per-model spec line. The agent NEEDS this — without it,
|
|
183
|
+
// it has to guess `supported_resolutions`/`supported_durations` and
|
|
184
|
+
// either invents values (then the API silently substitutes) or asks
|
|
185
|
+
// the user to clarify what's only knowable from this list.
|
|
186
|
+
//
|
|
187
|
+
// Rendering rule: emit a line for EVERY known constraint that is
|
|
188
|
+
// applicable for this model's type — even when the value is 0 / null.
|
|
189
|
+
// Hiding "0 cap" lines used to mean the agent couldn't distinguish
|
|
190
|
+
// "this model rejects DNA" (cap = 0) from "I don't know" (field
|
|
191
|
+
// missing). Now an explicit `max_dna: 0 (DNA not supported)` says the
|
|
192
|
+
// model says no, and absence means the API doesn't expose the field.
|
|
193
|
+
const formatSpecs = m => {
|
|
194
|
+
const parts = [];
|
|
195
|
+
if (m.haveThinking && Array.isArray(m.thinkingLevels) && m.thinkingLevels.length) {
|
|
196
|
+
parts.push(`thinking_level: ${m.thinkingLevels.map(level => level.id).join('/')} (default ${m.thinkingDefault})`);
|
|
197
|
+
}
|
|
198
|
+
const types = Array.isArray(m.types) ? m.types : [];
|
|
199
|
+
const isVideoType = types.some(t =>
|
|
200
|
+
['text_to_video', 'img_to_video', 'video_to_video', 'elements',
|
|
201
|
+
'firstlastgenerations', 'lipsync-image', 'lipsync-video', 'draw_to_video'].includes(t)
|
|
202
|
+
);
|
|
203
|
+
const isElements = types.includes('elements');
|
|
204
|
+
const isV2V = types.includes('video_to_video');
|
|
205
|
+
const isLipsyncVideo = types.includes('lipsync-video');
|
|
206
|
+
const isLipsyncImage = types.includes('lipsync-image');
|
|
207
|
+
const isImageEdit = types.includes('image_editing');
|
|
208
|
+
const isImage = types.includes('text_to_img') || isImageEdit;
|
|
209
|
+
|
|
210
|
+
if (Array.isArray(m.supported_resolutions) && m.supported_resolutions.length) {
|
|
211
|
+
const mult = m.resolution_multipliers || {};
|
|
212
|
+
parts.push(
|
|
213
|
+
'resolutions: ' +
|
|
214
|
+
m.supported_resolutions
|
|
215
|
+
.map(r => (mult[r] != null && mult[r] !== 1 ? `${r} (${mult[r]}×)` : r))
|
|
216
|
+
.join(' · ')
|
|
217
|
+
);
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
if (m.video_input_credit != null) {
|
|
221
|
+
const vm = m.video_input_resolution_multipliers || {};
|
|
222
|
+
const tiers = Object.keys(vm).length
|
|
223
|
+
? ' · ' + Object.entries(vm).map(([r, mult]) => `${r} (${mult}×)`).join(' · ')
|
|
224
|
+
: '';
|
|
225
|
+
parts.push(`video_input_price: ${m.video_input_credit} credits/combined-second${tiers} · bill nominal input + output seconds (≤0.15s encoder padding snaps)`);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// Output durations (video gen output, not source video)
|
|
229
|
+
if (Array.isArray(m.supported_durations) && m.supported_durations.length) {
|
|
230
|
+
const ds = m.supported_durations;
|
|
231
|
+
const sorted = [...ds].sort((a, b) => a - b);
|
|
232
|
+
const isRange = sorted.length > 2 && sorted.every((v, i) => i === 0 || v - sorted[i - 1] === 1);
|
|
233
|
+
parts.push(`durations: ${isRange ? `${sorted[0]}-${sorted[sorted.length - 1]}s` : sorted.join('/') + 's'}`);
|
|
234
|
+
} else if (isVideoType && (m.min_output_duration != null || m.max_output_duration != null)) {
|
|
235
|
+
parts.push(`duration_range: ${m.min_output_duration ?? '?'}-${m.max_output_duration ?? '?'}s${m.default_duration != null ? ` (default ${m.default_duration}s)` : ''}`);
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
// Aspect ratios — prefer per-type override if set
|
|
239
|
+
const ratios = m.supported_aspect_ratios_by_type
|
|
240
|
+
? Object.entries(m.supported_aspect_ratios_by_type).map(([t, arr]) => `${t}: ${arr.join('/')}`)
|
|
241
|
+
: null;
|
|
242
|
+
if (ratios) {
|
|
243
|
+
parts.push(`aspect (per-type): ${ratios.join(' | ')}`);
|
|
244
|
+
} else if (Array.isArray(m.supported_aspect_ratios) && m.supported_aspect_ratios.length) {
|
|
245
|
+
parts.push(`aspect: ${m.supported_aspect_ratios.join(', ')}${m.default_aspect_ratio ? ` (default ${m.default_aspect_ratio})` : ''}`);
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
// Reference-input caps — show the slot relevant for this model family.
|
|
249
|
+
// The same conceptual "max reference images" lives under THREE field
|
|
250
|
+
// names depending on the model type. Be explicit about which is which
|
|
251
|
+
// so the agent reads the right one.
|
|
252
|
+
if (isImage || isImageEdit) {
|
|
253
|
+
parts.push(`max_reference_images: ${m.max_reference_images ?? 0}${(m.max_reference_images ?? 0) === 0 ? ' (no refs)' : ''}`);
|
|
254
|
+
}
|
|
255
|
+
if (isElements) {
|
|
256
|
+
parts.push(`elements caps: imgs=${m.elements_max_images ?? 0} · vids=${m.elements_max_videos ?? 0} · audio=${m.elements_max_audio ?? 0}`);
|
|
257
|
+
}
|
|
258
|
+
if (isV2V) {
|
|
259
|
+
parts.push(`v2v ref caps: imgs=${m.max_images ?? 0} · vids=${m.max_videos ?? 0} · elements=${m.max_elements ?? 0} · audio=${m.max_audio ?? 0}`);
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// Visual DNA cap — always show for image / elements / video, even if 0.
|
|
263
|
+
// Use the authoritative supports_visual_dna flag when available; fall
|
|
264
|
+
// back to inferring from cap > 0 for older API responses.
|
|
265
|
+
const dnaSupported = typeof m.supports_visual_dna === 'boolean'
|
|
266
|
+
? m.supports_visual_dna
|
|
267
|
+
: (m.max_visual_dna ?? 0) > 0;
|
|
268
|
+
if (isImage || isVideoType) {
|
|
269
|
+
const cap = m.max_visual_dna;
|
|
270
|
+
if (dnaSupported && cap != null && cap > 0) parts.push(`max_visual_dna: ${cap}`);
|
|
271
|
+
else if (dnaSupported && cap == null) parts.push('visual_dna: supported (no cap published — confirm before passing >3)');
|
|
272
|
+
else parts.push('visual_dna: not supported');
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
// Source-video duration constraints — only matter for tools that take
|
|
276
|
+
// an INPUT video (lipsync-video, video_to_video).
|
|
277
|
+
if (isLipsyncVideo || isV2V) {
|
|
278
|
+
if (m.min_video_duration != null || m.max_video_duration != null) {
|
|
279
|
+
parts.push(`source_video: ${m.min_video_duration ?? '?'}-${m.max_video_duration ?? '?'}s`);
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
// Audio input — lipsync, elements, music-driven flows.
|
|
284
|
+
if (m.max_audio_duration != null || m.min_audio_duration != null) {
|
|
285
|
+
parts.push(`audio_input: ${m.min_audio_duration ?? '?'}-${m.max_audio_duration ?? '?'}s${m.audio_max_follows_video_duration ? ' (max follows video)' : ''}`);
|
|
286
|
+
}
|
|
287
|
+
if (Array.isArray(m.supported_audio_formats) && m.supported_audio_formats.length) {
|
|
288
|
+
parts.push(`audio_formats: ${m.supported_audio_formats.join('/')}`);
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
// Native sound generation (video models that emit synced audio)
|
|
292
|
+
if (m.sound_generation_type === 'native') {
|
|
293
|
+
const mult = m.sound_credit_multiplier && m.sound_credit_multiplier !== 1
|
|
294
|
+
? ` (${m.sound_credit_multiplier}×)`
|
|
295
|
+
: '';
|
|
296
|
+
parts.push(`sound: native${mult}${m.sound_enabled_by_default ? ' on-by-default' : ''}`);
|
|
297
|
+
} else if (m.sound_baked_in) {
|
|
298
|
+
parts.push('sound: baked-in (always on; type=none hides the toggle — still real audio)');
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
// Prompt constraints
|
|
302
|
+
if (m.requires_prompt === false) parts.push('prompt: optional');
|
|
303
|
+
if (m.min_prompt_length != null || m.max_prompt_length != null) {
|
|
304
|
+
parts.push(`prompt_length: ${m.min_prompt_length ?? 0}-${m.max_prompt_length ?? '∞'} chars`);
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
// Upload cap (when present)
|
|
308
|
+
if (m.max_file_size != null) {
|
|
309
|
+
const mb = Math.round(m.max_file_size / (1024 * 1024));
|
|
310
|
+
parts.push(`max_file_size: ${mb}MB`);
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
// Images-per-request (Midjourney-style fixed-N output)
|
|
314
|
+
if (m.images_per_request != null && m.images_per_request !== 1) {
|
|
315
|
+
parts.push(`images_per_request: ${m.images_per_request}`);
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
// Quality tiers (image models that support quality selection)
|
|
319
|
+
if (Array.isArray(m.supported_qualities) && m.supported_qualities.length) {
|
|
320
|
+
const qMult = m.quality_multipliers || {};
|
|
321
|
+
const qParts = m.supported_qualities.map(q =>
|
|
322
|
+
qMult[q] && qMult[q] !== 1 ? `${q}(${qMult[q]}×)` : q
|
|
323
|
+
);
|
|
324
|
+
parts.push(`quality: ${qParts.join(' · ')}${m.default_quality ? ` (default ${m.default_quality})` : ''}`);
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
// Fixed-price override (some models charge a flat rate per resolution instead of per-second)
|
|
328
|
+
if (m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object' && Object.keys(m.flat_credit_by_resolution).length) {
|
|
329
|
+
const fp = Object.entries(m.flat_credit_by_resolution).map(([k, v]) => `${k}:${v}cr`).join(' · ');
|
|
330
|
+
parts.push(`flat_price: ${fp}`);
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
// Estimated generation time (wall-clock at base settings)
|
|
334
|
+
if (m.estimated_duration_seconds != null) {
|
|
335
|
+
parts.push(`est_time: ~${m.estimated_duration_seconds}s`);
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
// NSFW flag
|
|
339
|
+
if (m.nsfw_only) {
|
|
340
|
+
parts.push('nsfw: required');
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
return parts.length ? `\n ${parts.join(' | ')}` : '';
|
|
344
|
+
};
|
|
345
|
+
|
|
346
|
+
// The FULL catalog with every spec line measured 140,590 chars — past what
|
|
347
|
+
// hosts accept, on the one discovery tool the skill tells the model to call
|
|
348
|
+
// when it is unsure. Unfiltered, emit the one-line form (enough to choose a
|
|
349
|
+
// model); once `type` narrows it, the set is small enough for full specs.
|
|
350
|
+
const detailed = !!type;
|
|
351
|
+
// Summaries run to a paragraph each; across the whole catalog that alone
|
|
352
|
+
// is most of the payload. Unfiltered, one clause is enough to choose by.
|
|
353
|
+
const brief = (s) => {
|
|
354
|
+
if (!s) return '';
|
|
355
|
+
const flat = String(s).replace(/\s+/g, ' ').trim();
|
|
356
|
+
return flat.length > 130 ? flat.slice(0, 127).trimEnd() + '…' : flat;
|
|
357
|
+
};
|
|
358
|
+
// Text models bill per token — the flat `credit` is not what the user pays,
|
|
359
|
+
// so show the real per-1K rates when the API supplies them. Without this the
|
|
360
|
+
// "cheapest model that fits" rule is unusable for chat.
|
|
361
|
+
//
|
|
362
|
+
// Video-type models are the same problem in a different shape: kolbo-api's
|
|
363
|
+
// credit engine (credManagment.js) treats "charge per second of requested
|
|
364
|
+
// duration" as the UNIVERSAL rule for any type in
|
|
365
|
+
// [video, firstlast, elements, motion_graphic, cast] — not a per-model
|
|
366
|
+
// exception, the default. So `credit: 9` on a model with duration 8 is
|
|
367
|
+
// really 72 credits, and nothing in the catalog said so: an agent quoting
|
|
368
|
+
// cost from the bare `credit` field alone is wrong by exactly the
|
|
369
|
+
// requested duration, every time. `flat_credit_by_resolution` is the one
|
|
370
|
+
// carve-out — those models are charged the flat rate regardless of
|
|
371
|
+
// duration, so they're excluded here the same way credManagment.js
|
|
372
|
+
// excludes them (resolveFlatCredit wins over the multiplier).
|
|
373
|
+
const PER_SECOND_TYPES = ['video', 'firstlast', 'elements', 'motion_graphic', 'cast'];
|
|
374
|
+
const isPerSecondVideo = m => {
|
|
375
|
+
const types = Array.isArray(m.types) ? m.types : (m.type ? [m.type] : []);
|
|
376
|
+
const billedPerSecond = types.some(t => PER_SECOND_TYPES.some(kw => String(t).includes(kw)));
|
|
377
|
+
const hasFlatOverride = m.flat_credit_by_resolution && typeof m.flat_credit_by_resolution === 'object'
|
|
378
|
+
&& Object.keys(m.flat_credit_by_resolution).length > 0;
|
|
379
|
+
return billedPerSecond && !hasFlatOverride;
|
|
380
|
+
};
|
|
381
|
+
const cost = m => (m.output_token_rate != null
|
|
382
|
+
? `${m.input_token_rate ?? '?'}/${m.output_token_rate} credits per 1K tokens (in/out)`
|
|
383
|
+
: isPerSecondVideo(m)
|
|
384
|
+
? `${m.credit} credits/output-second${m.video_input_credit != null ? `; ${m.video_input_credit} credits/combined-second with video input` : ''}`
|
|
385
|
+
: `${m.credit} credits`);
|
|
386
|
+
const formatModel = m =>
|
|
387
|
+
`${m.identifier} (${m.name}) - ${cost(m)}${m.recommended ? ' [RECOMMENDED]' : ''}${m.new_model ? ' [NEW]' : ''}${m.summary ? ` — ${detailed ? m.summary : brief(m.summary)}` : ''}${detailed ? formatSpecs(m) : ''}`;
|
|
388
|
+
|
|
389
|
+
const sections = [];
|
|
390
|
+
|
|
391
|
+
if (!detailed) {
|
|
392
|
+
// The catalog is ~428 models. Listing all of them is both far past the
|
|
393
|
+
// text budget AND useless to choose from — so unfiltered, surface the
|
|
394
|
+
// curated picks and make the model narrow by `type` for the rest. This
|
|
395
|
+
// matches the connector rule of steering to a CONCRETE model.
|
|
396
|
+
const picks = result.models.filter(m => m.recommended || m.new_model);
|
|
397
|
+
if (picks.length) {
|
|
398
|
+
sections.push(`Recommended & new (${picks.length}):\n${picks.map(formatModel).join('\n')}`);
|
|
399
|
+
}
|
|
400
|
+
const text = `Kolbo model catalog — ${result.count} models total.\n\n`
|
|
401
|
+
+ `${sections.join('\n\n')}\n\n`
|
|
402
|
+
+ 'This shortlist is BADGE-BASED (recommended/new) — it is not a recommendation to '
|
|
403
|
+
+ 'use the newest or biggest model. To pick properly, re-call with `type` and choose by '
|
|
404
|
+
+ 'each model\'s strengths summary, taking the cheapest one that covers the task. '
|
|
405
|
+
+ 'To see everything in a '
|
|
406
|
+
+ 'category (with per-model resolutions, durations, aspect ratios and reference-image '
|
|
407
|
+
+ 'caps), re-call with `type`:\n'
|
|
408
|
+
+ ' text_to_img · image_editing · text_to_video · img_to_video · video_to_video ·\n'
|
|
409
|
+
+ ' first_last_frame · elements · lipsync · music_gen · text_to_speech ·\n'
|
|
410
|
+
+ ' text_to_sound · stt · three_d · text\n\n'
|
|
411
|
+
+ 'Use the "identifier" value as the "model" parameter in generate tools. '
|
|
412
|
+
+ 'For EVERY model + its exact identifier, re-call with format: "json" (compact index of the '
|
|
413
|
+
+ 'whole catalog). Add `type` to that call for the full raw documents with all caps.';
|
|
414
|
+
return respond(text);
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
if (withSummary.length > 0) {
|
|
418
|
+
sections.push(`Auto-selectable models (${withSummary.length}) — If the user already named a model/family (this turn or earlier), use that family — do not cheapest-swap. Otherwise CHOOSE BY THE SUMMARY after each "—": match it to what the user asked for, then take the CHEAPEST model that fits. Credit cost, [NEW] and [RECOMMENDED] are not reasons to pick a model:\n${withSummary.map(formatModel).join('\n')}`);
|
|
419
|
+
}
|
|
420
|
+
if (withoutSummary.length > 0) {
|
|
421
|
+
sections.push(`Named-only models (${withoutSummary.length}) — only use if the user explicitly requests by name:\n${withoutSummary.map(formatModel).join('\n')}`);
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
const text = `Available ${type} models (${result.count}):\n\n${sections.join('\n\n')}\n\nEvery ${type} model in the catalog is listed above — both sections together are the complete set. Use the "identifier" value as the "model" parameter in generate tools. For the raw documents (programmatic cap validation), re-call with format: "json".`;
|
|
425
|
+
return respond(text);
|
|
426
|
+
}
|
|
427
|
+
);
|
|
428
|
+
|
|
429
|
+
// ─── check_credits ─────────────────────────────────────────
|
|
430
|
+
server.tool(
|
|
431
|
+
'check_credits',
|
|
432
|
+
'Check your remaining Kolbo credit balance.',
|
|
433
|
+
{},
|
|
434
|
+
async () => {
|
|
435
|
+
const result = await client.get('/v1/account/credits');
|
|
436
|
+
|
|
437
|
+
return {
|
|
438
|
+
content: [{
|
|
439
|
+
type: 'text',
|
|
440
|
+
text: `Credit Balance:\n- Total: ${result.credits.total}\n- Plan credits: ${result.credits.plan_credits}\n- Credit pack: ${result.credits.credit_pack}\n- Redemption: ${result.credits.redemption}`
|
|
441
|
+
}]
|
|
442
|
+
};
|
|
443
|
+
}
|
|
444
|
+
);
|
|
445
|
+
|
|
446
|
+
// ─── get_session_usage ─────────────────────────────────────
|
|
447
|
+
// Real, multiplier-adjusted credit spend tagged with the caller's
|
|
448
|
+
// X-Kolbo-Caller-Session-Id (set automatically by the parent process —
|
|
449
|
+
// no need to pass it). Use this to give the user an honest "you've spent
|
|
450
|
+
// X credits in this app session" instead of estimating from base credits.
|
|
451
|
+
server.tool(
|
|
452
|
+
'get_session_usage',
|
|
453
|
+
'Fetch real, multiplier-adjusted credit spend for the current Kolbo Code app session. Use when the user asks "how much did I spend?" or before/after a large bulk job so you can quote actual cost (not an estimate from base credits). Returns total + per-tool breakdown + per-model breakdown + a recent list. The caller-session-id is forwarded automatically by the MCP HTTP client. ONLY works when running under Kolbo Code — on the claude.ai connector and other hosts there is no per-app session to scope to; use check_credits there instead.',
|
|
454
|
+
{},
|
|
455
|
+
async () => {
|
|
456
|
+
// Session scoping needs KOLBO_CALLER_SESSION_ID, which only the Kolbo Code
|
|
457
|
+
// parent process sets. The remote connector serves every user from one
|
|
458
|
+
// process, so it is never present there — calling anyway just returns a 400
|
|
459
|
+
// telling the user to reconfigure a process they do not control.
|
|
460
|
+
if (!process.env.KOLBO_CALLER_SESSION_ID) {
|
|
461
|
+
return {
|
|
462
|
+
content: [{
|
|
463
|
+
type: 'text',
|
|
464
|
+
text: JSON.stringify({
|
|
465
|
+
unavailable: 'Per-session usage is only tracked when running under Kolbo Code.',
|
|
466
|
+
reason: 'This host does not scope tool calls to an app session, so there is no session to total.',
|
|
467
|
+
use_instead: 'check_credits for the current balance, or the Usage page at https://app.kolbo.ai.'
|
|
468
|
+
})
|
|
469
|
+
}]
|
|
470
|
+
};
|
|
471
|
+
}
|
|
472
|
+
try {
|
|
473
|
+
const r = await client.get('/credit-usage/by-caller-session');
|
|
474
|
+
// The endpoint returns { message, data: { total, count, by_tool, by_model, recent[] } }
|
|
475
|
+
return {
|
|
476
|
+
content: [{
|
|
477
|
+
type: 'text',
|
|
478
|
+
text: JSON.stringify(r.data || r, null, 2)
|
|
479
|
+
}]
|
|
480
|
+
};
|
|
481
|
+
} catch (err) {
|
|
482
|
+
// 400 from the endpoint means no caller-session-id was forwarded —
|
|
483
|
+
// surface a clear hint instead of a generic API error.
|
|
484
|
+
const hint = err?.status === 400
|
|
485
|
+
? 'No caller-session-id was forwarded. Ensure the parent process (Kolbo Code / desktop sidecar) sets KOLBO_CALLER_SESSION_ID in this MCP\'s env, or call again after at least one media generation has fired.'
|
|
486
|
+
: err?.message || 'Failed to fetch session usage';
|
|
487
|
+
return {
|
|
488
|
+
content: [{ type: 'text', text: JSON.stringify({ error: hint }, null, 2) }]
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
}
|
|
492
|
+
);
|
|
493
|
+
|
|
494
|
+
// ─── show_plans ──────────────────────────────────
|
|
495
|
+
// The upgrade card. Also rendered automatically when a generation is refused
|
|
496
|
+
// for credits — see insufficientCreditsResult() in _shared.js.
|
|
497
|
+
// Not registered at all under `commerce: false` (ChatGPT app profile): the
|
|
498
|
+
// OpenAI directory forbids selling digital goods, and this tool exists only
|
|
499
|
+
// to sell subscriptions and credit packs.
|
|
500
|
+
if (options.commerce === false) return;
|
|
501
|
+
server.tool(
|
|
502
|
+
'show_plans',
|
|
503
|
+
'Show the user their Kolbo credit balance, current plan, and the available upgrade plans / credit packs as an interactive card. Use when the user asks about pricing, plans, upgrading, or how to get more credits. Prices shown are live and promo-adjusted. The card links to app.kolbo.ai/pricing to complete a purchase — never quote prices from memory, and never claim to have made a purchase for them.',
|
|
504
|
+
{},
|
|
505
|
+
async () => {
|
|
506
|
+
const data = await client.get('/v1/account/plans');
|
|
507
|
+
const structured = {
|
|
508
|
+
widget: 'plans',
|
|
509
|
+
reason: 'requested',
|
|
510
|
+
balance: data?.credits?.total,
|
|
511
|
+
current_plan: data?.current_plan || null,
|
|
512
|
+
plans: data?.plans || [],
|
|
513
|
+
credit_packs: data?.credit_packs || [],
|
|
514
|
+
pricing_url: data?.pricing_url || 'https://app.kolbo.ai/pricing',
|
|
515
|
+
};
|
|
516
|
+
// structuredContent SHADOWS the text on widget hosts, so everything the
|
|
517
|
+
// agent needs to talk about pricing has to live in the object above.
|
|
518
|
+
const text = JSON.stringify({
|
|
519
|
+
credits: structured.balance,
|
|
520
|
+
current_plan: structured.current_plan,
|
|
521
|
+
plans: structured.plans,
|
|
522
|
+
credit_packs: structured.credit_packs,
|
|
523
|
+
pricing_url: structured.pricing_url,
|
|
524
|
+
_hint: 'A plans card is rendered for the user. Summarise briefly; do NOT paste the price table. Purchases are completed by the user on the pricing page — you cannot buy on their behalf.',
|
|
525
|
+
}, null, 2);
|
|
526
|
+
return uiResult(UI.plans, text, structured);
|
|
527
|
+
}
|
|
528
|
+
);
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
module.exports = { registerModelTools };
|