@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +367 -283
- package/dist/index.js +1 -1
- package/dist/model-cache.d.ts +22 -12
- package/dist/model-cache.js +58 -21
- package/dist/openrouter-api.d.ts +45 -0
- package/dist/openrouter-api.js +50 -0
- package/dist/tool-descriptions.d.ts +19 -0
- package/dist/tool-descriptions.js +573 -0
- package/dist/tool-handlers/analyze-audio.js +5 -1
- package/dist/tool-handlers/analyze-image.js +6 -5
- package/dist/tool-handlers/analyze-video.js +6 -5
- package/dist/tool-handlers/async-chat.d.ts +51 -0
- package/dist/tool-handlers/async-chat.js +216 -0
- package/dist/tool-handlers/audio-utils.js +4 -2
- package/dist/tool-handlers/chat-completion.js +1 -1
- package/dist/tool-handlers/fetch-utils.js +16 -2
- package/dist/tool-handlers/generate-audio.js +2 -4
- package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
- package/dist/tool-handlers/generate-image-dedicated.js +176 -0
- package/dist/tool-handlers/generate-image-input.d.ts +3 -0
- package/dist/tool-handlers/generate-image-input.js +38 -0
- package/dist/tool-handlers/generate-image.d.ts +13 -51
- package/dist/tool-handlers/generate-image.js +32 -119
- package/dist/tool-handlers/generate-video.d.ts +2 -2
- package/dist/tool-handlers/generate-video.js +78 -30
- package/dist/tool-handlers/image-utils.d.ts +1 -0
- package/dist/tool-handlers/image-utils.js +26 -16
- package/dist/tool-handlers/openrouter-errors.js +6 -2
- package/dist/tool-handlers/path-safety.js +32 -5
- package/dist/tool-handlers/provider-routing.js +7 -2
- package/dist/tool-handlers/rerank.js +2 -5
- package/dist/tool-handlers/search-models.d.ts +2 -2
- package/dist/tool-handlers/search-models.js +2 -6
- package/dist/tool-handlers/speech-to-text.d.ts +20 -0
- package/dist/tool-handlers/speech-to-text.js +140 -0
- package/dist/tool-handlers/structured-output.d.ts +8 -0
- package/dist/tool-handlers/structured-output.js +11 -0
- package/dist/tool-handlers/text-to-speech.d.ts +29 -0
- package/dist/tool-handlers/text-to-speech.js +105 -0
- package/dist/tool-handlers/video-utils.js +6 -9
- package/dist/tool-handlers.js +253 -125
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +27 -15
package/dist/tool-handlers.js
CHANGED
|
@@ -14,6 +14,11 @@ import { handleAnalyzeVideo } from './tool-handlers/analyze-video.js';
|
|
|
14
14
|
import { handleGenerateVideo, handleGetVideoStatus, handleGenerateVideoFromImage, } from './tool-handlers/generate-video.js';
|
|
15
15
|
import { handleRerankDocuments } from './tool-handlers/rerank.js';
|
|
16
16
|
import { handleHealthCheck } from './tool-handlers/health-check.js';
|
|
17
|
+
import { handleGenerateImageDedicated, } from './tool-handlers/generate-image-dedicated.js';
|
|
18
|
+
import { handleTextToSpeech } from './tool-handlers/text-to-speech.js';
|
|
19
|
+
import { handleSpeechToText } from './tool-handlers/speech-to-text.js';
|
|
20
|
+
import { handleStartChatCompletion, handleGetChatCompletionStatus, } from './tool-handlers/async-chat.js';
|
|
21
|
+
import { TOOL_DESCRIPTIONS } from './tool-descriptions.js';
|
|
17
22
|
function wrapToolArgs(a) {
|
|
18
23
|
return { params: { arguments: a ?? {} } };
|
|
19
24
|
}
|
|
@@ -49,124 +54,9 @@ function buildProgressHook(server, progressToken) {
|
|
|
49
54
|
};
|
|
50
55
|
}
|
|
51
56
|
function extractProgressToken(req) {
|
|
52
|
-
const meta = req?.params
|
|
53
|
-
?._meta;
|
|
57
|
+
const meta = req?.params?._meta;
|
|
54
58
|
return meta?.progressToken;
|
|
55
59
|
}
|
|
56
|
-
// ---------------------------------------------------------------------------
|
|
57
|
-
// Tool descriptions include explicit "Fails when" and "Works with" sections
|
|
58
|
-
// per arxiv 2602.18764 (Schema-Guided Dialogue / MCP convergence). Explicit
|
|
59
|
-
// failure-mode documentation reduces misrouted calls and helps the model
|
|
60
|
-
// pick the right recovery path after an error.
|
|
61
|
-
const TOOL_DESCRIPTIONS = {
|
|
62
|
-
chat_completion: 'Send messages to an OpenRouter model and get a text response. Supports provider routing ' +
|
|
63
|
-
'(quantizations / ignore / sort / order / require_parameters / data_collection / allow_fallbacks), ' +
|
|
64
|
-
'model variant suffixes (`:nitro` fastest, `:floor` cheapest, `:exacto` tool-calling accuracy), ' +
|
|
65
|
-
'reasoning token passthrough, web search, and response caching.\n\n' +
|
|
66
|
-
'Fails when:\n' +
|
|
67
|
-
'- INVALID_INPUT: messages array is empty\n' +
|
|
68
|
-
'- UPSTREAM_REFUSED: provider rejected the request (credits, content policy, or rate limit)\n' +
|
|
69
|
-
'- UPSTREAM_TIMEOUT: upstream did not respond within the SDK timeout\n' +
|
|
70
|
-
'- MODEL_NOT_FOUND: model slug does not exist on OpenRouter\n\n' +
|
|
71
|
-
'Works with: validate_model (pre-flight model id check), search_models (discover models).',
|
|
72
|
-
analyze_image: 'Analyze an image with a vision model. Accepts local file paths, http(s) URLs, or base64 data URLs. ' +
|
|
73
|
-
'Output is model-generated and tagged `_meta.content_is_untrusted: true`.\n\n' +
|
|
74
|
-
'Fails when:\n' +
|
|
75
|
-
'- INVALID_INPUT: image_path missing or malformed\n' +
|
|
76
|
-
'- UNSAFE_PATH: local path escaped the sandbox\n' +
|
|
77
|
-
'- RESOURCE_TOO_LARGE: image exceeded configured fetch size cap\n' +
|
|
78
|
-
'- UPSTREAM_REFUSED: provider SSRF guard blocked the URL or content policy rejected\n\n' +
|
|
79
|
-
'Works with: search_models (find vision-capable models), generate_image (follow-up creation).',
|
|
80
|
-
analyze_audio: 'Transcribe or analyze an audio file (WAV / MP3 / FLAC / OGG / etc.) using a multimodal model. ' +
|
|
81
|
-
'Output is tagged `_meta.content_is_untrusted: true`.\n\n' +
|
|
82
|
-
'Fails when:\n' +
|
|
83
|
-
'- INVALID_INPUT: audio_path missing\n' +
|
|
84
|
-
'- UNSUPPORTED_FORMAT: decoder could not identify the file as audio\n' +
|
|
85
|
-
'- RESOURCE_TOO_LARGE: input exceeded size cap\n' +
|
|
86
|
-
'- UPSTREAM_REFUSED: blocked host or content policy\n\n' +
|
|
87
|
-
'Works with: generate_audio (text-to-speech follow-up).',
|
|
88
|
-
analyze_video: 'Describe or analyze a video (mp4 / mpeg / mov / webm) using a multimodal model. Default model: ' +
|
|
89
|
-
'google/gemini-2.5-flash. Output is tagged `_meta.content_is_untrusted: true`.\n\n' +
|
|
90
|
-
'Fails when:\n' +
|
|
91
|
-
'- INVALID_INPUT: video_path missing\n' +
|
|
92
|
-
'- UNSUPPORTED_FORMAT: not a recognized video container\n' +
|
|
93
|
-
'- RESOURCE_TOO_LARGE: exceeds fetch cap\n' +
|
|
94
|
-
'- UPSTREAM_REFUSED: SSRF block or provider refusal\n\n' +
|
|
95
|
-
'Works with: generate_video (text-to-video), get_video_status (poll async jobs).',
|
|
96
|
-
search_models: 'Search OpenRouter\'s model catalog by name, provider, or capability. Returns a paginated list; ' +
|
|
97
|
-
'use `offset` / `limit` / `next_offset` to page through.\n\n' +
|
|
98
|
-
'Fails when:\n' +
|
|
99
|
-
'- UPSTREAM_HTTP: /models endpoint returned an error\n' +
|
|
100
|
-
'- UPSTREAM_REFUSED: invalid API key\n\n' +
|
|
101
|
-
'Works with: validate_model, get_model_info.',
|
|
102
|
-
get_model_info: 'Get pricing / context-length / capability details for a specific model id.\n\n' +
|
|
103
|
-
'Fails when:\n' +
|
|
104
|
-
'- INVALID_INPUT: model not provided\n' +
|
|
105
|
-
'- MODEL_NOT_FOUND: model slug does not exist\n' +
|
|
106
|
-
'- UPSTREAM_HTTP: model list fetch failed\n\n' +
|
|
107
|
-
'Works with: search_models (discover ids), validate_model (cheap existence check).',
|
|
108
|
-
validate_model: 'Check whether a model id exists on OpenRouter. Cheap boolean lookup against the cached catalog.\n\n' +
|
|
109
|
-
'Fails when:\n' +
|
|
110
|
-
'- INVALID_INPUT: model not provided\n' +
|
|
111
|
-
'- UPSTREAM_HTTP: catalog refresh failed\n\n' +
|
|
112
|
-
'Works with: get_model_info (detailed lookup), chat_completion (pre-flight validation).',
|
|
113
|
-
generate_image: 'Generate an image from a text prompt. Optional reference images condition the output for style / ' +
|
|
114
|
-
'identity consistency. Default model: google/gemini-2.5-flash-image.\n\n' +
|
|
115
|
-
'Fails when:\n' +
|
|
116
|
-
'- INVALID_INPUT: prompt empty, bad aspect_ratio / image_size, unreadable reference image\n' +
|
|
117
|
-
'- UNSAFE_PATH: save_path or input_images path escaped the sandbox\n' +
|
|
118
|
-
'- UPSTREAM_REFUSED: provider content policy rejected or insufficient credits\n' +
|
|
119
|
-
'- MODEL_NOT_FOUND: model slug invalid\n\n' +
|
|
120
|
-
'Works with: analyze_image (verify the result), generate_video_from_image (next step in workflow).',
|
|
121
|
-
generate_audio: 'Generate audio (speech / music) from a text prompt. Format auto-detected, extension auto-corrected.\n\n' +
|
|
122
|
-
'Fails when:\n' +
|
|
123
|
-
'- INVALID_INPUT: prompt empty\n' +
|
|
124
|
-
'- UNSAFE_PATH: save_path escaped the sandbox\n' +
|
|
125
|
-
'- UPSTREAM_REFUSED: content policy or credit issues\n\n' +
|
|
126
|
-
'Works with: analyze_audio (verify the result).',
|
|
127
|
-
generate_video: 'Generate a video from a text prompt (optionally conditioned on first/last-frame or reference images). ' +
|
|
128
|
-
'Submits an async job, polls until completion or max_wait_ms, and downloads the result. Emits MCP ' +
|
|
129
|
-
'progress notifications when the client provides a `progressToken`. Default model: google/veo-3.1.\n\n' +
|
|
130
|
-
'Fails when:\n' +
|
|
131
|
-
'- INVALID_INPUT: prompt empty\n' +
|
|
132
|
-
'- UNSAFE_PATH: save_path or reference image paths escaped the sandbox\n' +
|
|
133
|
-
'- UPSTREAM_REFUSED: content policy, credits, or bad request\n' +
|
|
134
|
-
'- JOB_FAILED: provider marked the job as failed\n' +
|
|
135
|
-
'- UNSUPPORTED_FORMAT: reference/frame image could not be decoded\n\n' +
|
|
136
|
-
'Returns successfully with `_meta.code: JOB_STILL_RUNNING` (NOT an error) when the timeout ' +
|
|
137
|
-
'elapses — the response carries `_meta.video_id` so callers can resume via get_video_status.\n\n' +
|
|
138
|
-
'Works with: get_video_status (resume timed-out jobs), generate_video_from_image (narrower image-to-video variant).',
|
|
139
|
-
generate_video_from_image: 'Narrower convenience wrapper around generate_video for image-to-video workflows. Takes a single ' +
|
|
140
|
-
'`image` argument (used as the first frame) and `prompt`. Per arxiv 2511.03497, narrower tools with ' +
|
|
141
|
-
'fewer parameters improve tool-call hit rate. For last-frame conditioning or reference images, use ' +
|
|
142
|
-
'generate_video directly.\n\n' +
|
|
143
|
-
'Fails when:\n' +
|
|
144
|
-
'- INVALID_INPUT: image or prompt missing\n' +
|
|
145
|
-
'- UNSAFE_PATH: image path escaped the sandbox\n' +
|
|
146
|
-
'- UPSTREAM_REFUSED / JOB_FAILED: same as generate_video\n\n' +
|
|
147
|
-
'Returns successfully with `_meta.code: JOB_STILL_RUNNING` on timeout (resumable via ' +
|
|
148
|
-
'get_video_status).\n\n' +
|
|
149
|
-
'Works with: generate_video (full parameter surface), get_video_status.',
|
|
150
|
-
get_video_status: 'Poll an async video-generation job by id. Downloads the result when complete (and saves if save_path given).\n\n' +
|
|
151
|
-
'Fails when:\n' +
|
|
152
|
-
'- INVALID_INPUT: video_id missing\n' +
|
|
153
|
-
'- UNSAFE_PATH: save_path escaped the sandbox\n' +
|
|
154
|
-
'- JOB_FAILED: provider marked the job as failed\n\n' +
|
|
155
|
-
'Returns successfully with `_meta.code: JOB_STILL_RUNNING` (NOT an error) when the job is still ' +
|
|
156
|
-
'in flight — response carries `_meta.last_status` and `_meta.progress` so callers can retry later.\n\n' +
|
|
157
|
-
'Works with: generate_video, generate_video_from_image.',
|
|
158
|
-
rerank_documents: 'Re-order a list of documents by relevance to a query using an OpenRouter reranker. Default model: ' +
|
|
159
|
-
'cohere/rerank-english-v3.0.\n\n' +
|
|
160
|
-
'Fails when:\n' +
|
|
161
|
-
'- INVALID_INPUT: query missing, documents empty, non-string document elements\n' +
|
|
162
|
-
'- MODEL_NOT_FOUND: reranker model slug does not exist\n' +
|
|
163
|
-
'- UPSTREAM_HTTP: provider returned an error\n\n' +
|
|
164
|
-
'Works with: search_models (discover rerankers), chat_completion (answer grounded in top-ranked docs).',
|
|
165
|
-
health_check: 'Verify API-key validity, OpenRouter reachability, and return server + protocol versions. No args.\n\n' +
|
|
166
|
-
'Fails when: never returns an error result — always returns `{ ok, api_key_valid, ... }` so ops can ' +
|
|
167
|
-
'programmatically branch on the payload.\n\n' +
|
|
168
|
-
'Works with: every other tool (run once at startup to confirm credentials).',
|
|
169
|
-
};
|
|
170
60
|
export class ToolHandlers {
|
|
171
61
|
openai;
|
|
172
62
|
modelCache = ModelCache.getInstance();
|
|
@@ -202,8 +92,9 @@ export class ToolHandlers {
|
|
|
202
92
|
model: {
|
|
203
93
|
type: 'string',
|
|
204
94
|
description: 'Model ID (optional, uses default). Append `:nitro` for the fastest variant, ' +
|
|
205
|
-
'`:floor` for the cheapest,
|
|
206
|
-
'
|
|
95
|
+
'`:floor` for the cheapest, `:free` for the free tier, `:online` for web search, ' +
|
|
96
|
+
'or `:exacto` for the best tool-calling accuracy. ' +
|
|
97
|
+
'Example: `openai/gpt-4o:nitro`. Or pass `online: true` for programmatic web search control.',
|
|
207
98
|
},
|
|
208
99
|
messages: {
|
|
209
100
|
type: 'array',
|
|
@@ -252,11 +143,11 @@ export class ToolHandlers {
|
|
|
252
143
|
},
|
|
253
144
|
include_reasoning: {
|
|
254
145
|
type: 'boolean',
|
|
255
|
-
description:
|
|
146
|
+
description: "Surface the model's chain-of-thought on `_meta.reasoning` for R1 / Opus 4.7 / Gemini Thinking.",
|
|
256
147
|
},
|
|
257
148
|
online: {
|
|
258
149
|
type: 'boolean',
|
|
259
|
-
description:
|
|
150
|
+
description: "Enable OpenRouter's web-search plugin (Exa-backed, $4 / 1000 results).",
|
|
260
151
|
},
|
|
261
152
|
web_max_results: {
|
|
262
153
|
type: 'number',
|
|
@@ -280,6 +171,63 @@ export class ToolHandlers {
|
|
|
280
171
|
required: ['messages'],
|
|
281
172
|
},
|
|
282
173
|
},
|
|
174
|
+
{
|
|
175
|
+
name: 'start_chat_completion',
|
|
176
|
+
description: TOOL_DESCRIPTIONS.start_chat_completion,
|
|
177
|
+
annotations: {
|
|
178
|
+
title: 'Start async chat completion',
|
|
179
|
+
readOnlyHint: false,
|
|
180
|
+
destructiveHint: false,
|
|
181
|
+
idempotentHint: false,
|
|
182
|
+
openWorldHint: true,
|
|
183
|
+
},
|
|
184
|
+
inputSchema: {
|
|
185
|
+
type: 'object',
|
|
186
|
+
properties: {
|
|
187
|
+
model: { type: 'string', description: 'Model ID (same as chat_completion).' },
|
|
188
|
+
messages: {
|
|
189
|
+
type: 'array',
|
|
190
|
+
minItems: 1,
|
|
191
|
+
items: {
|
|
192
|
+
type: 'object',
|
|
193
|
+
properties: {
|
|
194
|
+
role: { type: 'string', enum: ['system', 'user', 'assistant'] },
|
|
195
|
+
content: { oneOf: [{ type: 'string' }, { type: 'array', items: { type: 'object' } }] },
|
|
196
|
+
},
|
|
197
|
+
required: ['role', 'content'],
|
|
198
|
+
},
|
|
199
|
+
},
|
|
200
|
+
temperature: { type: 'number', minimum: 0, maximum: 2 },
|
|
201
|
+
max_tokens: { type: 'number', minimum: 1 },
|
|
202
|
+
provider: { type: 'object' },
|
|
203
|
+
include_reasoning: { type: 'boolean' },
|
|
204
|
+
online: { type: 'boolean' },
|
|
205
|
+
web_max_results: { type: 'number', minimum: 1 },
|
|
206
|
+
cache: { type: 'boolean' },
|
|
207
|
+
cache_ttl: { type: 'string' },
|
|
208
|
+
cache_clear: { type: 'boolean' },
|
|
209
|
+
},
|
|
210
|
+
required: ['messages'],
|
|
211
|
+
},
|
|
212
|
+
},
|
|
213
|
+
{
|
|
214
|
+
name: 'get_chat_completion_status',
|
|
215
|
+
description: TOOL_DESCRIPTIONS.get_chat_completion_status,
|
|
216
|
+
annotations: {
|
|
217
|
+
title: 'Get async chat completion status',
|
|
218
|
+
readOnlyHint: true,
|
|
219
|
+
destructiveHint: false,
|
|
220
|
+
idempotentHint: true,
|
|
221
|
+
openWorldHint: false,
|
|
222
|
+
},
|
|
223
|
+
inputSchema: {
|
|
224
|
+
type: 'object',
|
|
225
|
+
properties: {
|
|
226
|
+
job_id: { type: 'string', description: 'The job_id returned by start_chat_completion.' },
|
|
227
|
+
},
|
|
228
|
+
required: ['job_id'],
|
|
229
|
+
},
|
|
230
|
+
},
|
|
283
231
|
{
|
|
284
232
|
name: 'analyze_image',
|
|
285
233
|
description: TOOL_DESCRIPTIONS.analyze_image,
|
|
@@ -293,8 +241,16 @@ export class ToolHandlers {
|
|
|
293
241
|
inputSchema: {
|
|
294
242
|
type: 'object',
|
|
295
243
|
properties: {
|
|
296
|
-
image_path: {
|
|
297
|
-
|
|
244
|
+
image_path: {
|
|
245
|
+
type: 'string',
|
|
246
|
+
description: 'Required. Local path (inside OPENROUTER_INPUT_DIR sandbox), https URL, or data URL. ' +
|
|
247
|
+
'Good: `"photo.jpg"`. Bad: `"url": "..."` (wrong key), `"/etc/passwd"` (UNSAFE_PATH).',
|
|
248
|
+
},
|
|
249
|
+
question: {
|
|
250
|
+
type: 'string',
|
|
251
|
+
description: 'Optional question about the image. Defaults to "What\'s in this image?" if omitted. ' +
|
|
252
|
+
'Good: `"List all text"`. Bad: using `prompt` key (wrong name for this tool).',
|
|
253
|
+
},
|
|
298
254
|
model: { type: 'string' },
|
|
299
255
|
cache_input: {
|
|
300
256
|
type: 'boolean',
|
|
@@ -323,7 +279,8 @@ export class ToolHandlers {
|
|
|
323
279
|
properties: {
|
|
324
280
|
audio_path: {
|
|
325
281
|
type: 'string',
|
|
326
|
-
description: '
|
|
282
|
+
description: 'Local file path (sandboxed to OPENROUTER_INPUT_DIR / OPENROUTER_OUTPUT_DIR / cwd), ' +
|
|
283
|
+
'http(s) URL, or data URL (base64-encoded audio)',
|
|
327
284
|
},
|
|
328
285
|
question: {
|
|
329
286
|
type: 'string',
|
|
@@ -353,7 +310,8 @@ export class ToolHandlers {
|
|
|
353
310
|
properties: {
|
|
354
311
|
video_path: {
|
|
355
312
|
type: 'string',
|
|
356
|
-
description: '
|
|
313
|
+
description: 'Local file path (sandboxed to OPENROUTER_INPUT_DIR / OPENROUTER_OUTPUT_DIR / cwd), ' +
|
|
314
|
+
'http(s) URL, or base64 data URL. Supported: mp4 / mpeg / mov / webm.',
|
|
357
315
|
},
|
|
358
316
|
question: { type: 'string' },
|
|
359
317
|
model: { type: 'string' },
|
|
@@ -498,6 +456,69 @@ export class ToolHandlers {
|
|
|
498
456
|
required: ['prompt'],
|
|
499
457
|
},
|
|
500
458
|
},
|
|
459
|
+
{
|
|
460
|
+
name: 'generate_image_dedicated',
|
|
461
|
+
description: TOOL_DESCRIPTIONS.generate_image_dedicated,
|
|
462
|
+
annotations: {
|
|
463
|
+
title: 'Generate image (dedicated API)',
|
|
464
|
+
readOnlyHint: false,
|
|
465
|
+
destructiveHint: false,
|
|
466
|
+
idempotentHint: false,
|
|
467
|
+
openWorldHint: true,
|
|
468
|
+
},
|
|
469
|
+
inputSchema: {
|
|
470
|
+
type: 'object',
|
|
471
|
+
properties: {
|
|
472
|
+
prompt: {
|
|
473
|
+
type: 'string',
|
|
474
|
+
description: 'Text prompt describing the image to generate.',
|
|
475
|
+
},
|
|
476
|
+
model: {
|
|
477
|
+
type: 'string',
|
|
478
|
+
description: 'Image model ID. Default: google/gemini-2.5-flash-image. Browse available at https://openrouter.ai/collections/image-models',
|
|
479
|
+
},
|
|
480
|
+
resolution: {
|
|
481
|
+
type: 'string',
|
|
482
|
+
enum: ['512', '0.5K', '1K', '2K', '4K'],
|
|
483
|
+
description: 'Normalized resolution tier. Provider maps to closest supported size.',
|
|
484
|
+
},
|
|
485
|
+
aspect_ratio: {
|
|
486
|
+
type: 'string',
|
|
487
|
+
description: 'Aspect ratio (e.g. "1:1", "16:9", "9:16", "4:3", "21:9").',
|
|
488
|
+
},
|
|
489
|
+
quality: {
|
|
490
|
+
type: 'string',
|
|
491
|
+
enum: ['auto', 'low', 'medium', 'high'],
|
|
492
|
+
description: 'Image quality level.',
|
|
493
|
+
},
|
|
494
|
+
output_format: {
|
|
495
|
+
type: 'string',
|
|
496
|
+
enum: ['png', 'jpeg', 'webp', 'svg'],
|
|
497
|
+
description: 'Output image format.',
|
|
498
|
+
},
|
|
499
|
+
n: {
|
|
500
|
+
type: 'number',
|
|
501
|
+
minimum: 1,
|
|
502
|
+
maximum: 10,
|
|
503
|
+
description: 'Number of images to generate (model-dependent, default 1).',
|
|
504
|
+
},
|
|
505
|
+
input_references: {
|
|
506
|
+
type: 'array',
|
|
507
|
+
items: { type: 'string' },
|
|
508
|
+
description: 'Reference images for image-to-image workflows. Each entry: local path, http(s) URL, or data URL.',
|
|
509
|
+
},
|
|
510
|
+
save_path: { type: 'string', description: 'Save generated image to this path.' },
|
|
511
|
+
provider: {
|
|
512
|
+
type: 'object',
|
|
513
|
+
description: 'Provider routing overrides (order, sort, allow_fallbacks, etc.).',
|
|
514
|
+
},
|
|
515
|
+
cache: { type: 'boolean' },
|
|
516
|
+
cache_ttl: { type: 'string' },
|
|
517
|
+
cache_clear: { type: 'boolean' },
|
|
518
|
+
},
|
|
519
|
+
required: ['prompt'],
|
|
520
|
+
},
|
|
521
|
+
},
|
|
501
522
|
{
|
|
502
523
|
name: 'generate_audio',
|
|
503
524
|
description: TOOL_DESCRIPTIONS.generate_audio,
|
|
@@ -520,6 +541,97 @@ export class ToolHandlers {
|
|
|
520
541
|
required: ['prompt'],
|
|
521
542
|
},
|
|
522
543
|
},
|
|
544
|
+
{
|
|
545
|
+
name: 'text_to_speech',
|
|
546
|
+
description: TOOL_DESCRIPTIONS.text_to_speech,
|
|
547
|
+
annotations: {
|
|
548
|
+
title: 'Text to speech (dedicated API)',
|
|
549
|
+
readOnlyHint: false,
|
|
550
|
+
destructiveHint: false,
|
|
551
|
+
idempotentHint: false,
|
|
552
|
+
openWorldHint: true,
|
|
553
|
+
},
|
|
554
|
+
inputSchema: {
|
|
555
|
+
type: 'object',
|
|
556
|
+
properties: {
|
|
557
|
+
input: {
|
|
558
|
+
type: 'string',
|
|
559
|
+
description: 'Text to convert to speech.',
|
|
560
|
+
},
|
|
561
|
+
model: {
|
|
562
|
+
type: 'string',
|
|
563
|
+
description: 'TTS model. Default: openai/gpt-4o-mini-tts-2025-12-15. Also: google/gemini-flash-tts, mistral/voxtral-mini-tts.',
|
|
564
|
+
},
|
|
565
|
+
voice: {
|
|
566
|
+
type: 'string',
|
|
567
|
+
description: 'Voice ID (model-specific). Default: alloy. OpenAI voices: alloy, echo, fable, onyx, nova, shimmer.',
|
|
568
|
+
},
|
|
569
|
+
response_format: {
|
|
570
|
+
type: 'string',
|
|
571
|
+
enum: ['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm'],
|
|
572
|
+
description: 'Output audio format. Default: mp3.',
|
|
573
|
+
},
|
|
574
|
+
speed: {
|
|
575
|
+
type: 'number',
|
|
576
|
+
minimum: 0.25,
|
|
577
|
+
maximum: 4.0,
|
|
578
|
+
description: 'Speed of speech (0.25 to 4.0). Default: 1.0.',
|
|
579
|
+
},
|
|
580
|
+
instructions: {
|
|
581
|
+
type: 'string',
|
|
582
|
+
description: 'Tone/style instructions (e.g. "speak in a warm, friendly tone"). OpenAI models only.',
|
|
583
|
+
},
|
|
584
|
+
save_path: { type: 'string', description: 'Save audio to this path.' },
|
|
585
|
+
cache: { type: 'boolean' },
|
|
586
|
+
cache_ttl: { type: 'string' },
|
|
587
|
+
cache_clear: { type: 'boolean' },
|
|
588
|
+
},
|
|
589
|
+
required: ['input'],
|
|
590
|
+
},
|
|
591
|
+
},
|
|
592
|
+
{
|
|
593
|
+
name: 'speech_to_text',
|
|
594
|
+
description: TOOL_DESCRIPTIONS.speech_to_text,
|
|
595
|
+
annotations: {
|
|
596
|
+
title: 'Speech to text (dedicated API)',
|
|
597
|
+
readOnlyHint: true,
|
|
598
|
+
destructiveHint: false,
|
|
599
|
+
idempotentHint: false,
|
|
600
|
+
openWorldHint: true,
|
|
601
|
+
},
|
|
602
|
+
inputSchema: {
|
|
603
|
+
type: 'object',
|
|
604
|
+
properties: {
|
|
605
|
+
audio_path: {
|
|
606
|
+
type: 'string',
|
|
607
|
+
description: 'Audio file: local path (sandboxed), http(s) URL, or base64 data URL. Formats: mp3, wav, flac, ogg, webm, mp4.',
|
|
608
|
+
},
|
|
609
|
+
model: {
|
|
610
|
+
type: 'string',
|
|
611
|
+
description: 'STT model. Default: openai/whisper-1. Also: openai/gpt-4o-transcribe, openai/gpt-4o-mini-transcribe.',
|
|
612
|
+
},
|
|
613
|
+
language: {
|
|
614
|
+
type: 'string',
|
|
615
|
+
description: 'ISO-639-1 language code (e.g. "en", "es", "fr"). Improves accuracy.',
|
|
616
|
+
},
|
|
617
|
+
response_format: {
|
|
618
|
+
type: 'string',
|
|
619
|
+
enum: ['json', 'text', 'srt', 'verbose_json', 'vtt'],
|
|
620
|
+
description: 'Output format for transcription. Default: json.',
|
|
621
|
+
},
|
|
622
|
+
temperature: {
|
|
623
|
+
type: 'number',
|
|
624
|
+
minimum: 0,
|
|
625
|
+
maximum: 1,
|
|
626
|
+
description: 'Sampling temperature for transcription (0-1).',
|
|
627
|
+
},
|
|
628
|
+
cache: { type: 'boolean' },
|
|
629
|
+
cache_ttl: { type: 'string' },
|
|
630
|
+
cache_clear: { type: 'boolean' },
|
|
631
|
+
},
|
|
632
|
+
required: ['audio_path'],
|
|
633
|
+
},
|
|
634
|
+
},
|
|
523
635
|
{
|
|
524
636
|
name: 'generate_video',
|
|
525
637
|
description: TOOL_DESCRIPTIONS.generate_video,
|
|
@@ -661,7 +773,13 @@ export class ToolHandlers {
|
|
|
661
773
|
models_cached: { type: 'number' },
|
|
662
774
|
error: { type: 'string' },
|
|
663
775
|
},
|
|
664
|
-
required: [
|
|
776
|
+
required: [
|
|
777
|
+
'ok',
|
|
778
|
+
'server_version',
|
|
779
|
+
'protocol_version',
|
|
780
|
+
'api_key_valid',
|
|
781
|
+
'models_cached',
|
|
782
|
+
],
|
|
665
783
|
},
|
|
666
784
|
},
|
|
667
785
|
],
|
|
@@ -679,6 +797,10 @@ export class ToolHandlers {
|
|
|
679
797
|
switch (name) {
|
|
680
798
|
case 'chat_completion':
|
|
681
799
|
return handleChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
|
|
800
|
+
case 'start_chat_completion':
|
|
801
|
+
return handleStartChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
|
|
802
|
+
case 'get_chat_completion_status':
|
|
803
|
+
return handleGetChatCompletionStatus(wrapToolArgs(args));
|
|
682
804
|
case 'analyze_image':
|
|
683
805
|
return handleAnalyzeImage(wrapToolArgs(args), this.openai, this.defaultModel);
|
|
684
806
|
case 'analyze_audio':
|
|
@@ -693,8 +815,14 @@ export class ToolHandlers {
|
|
|
693
815
|
return handleValidateModel(wrapToolArgs(args), this.modelCache, this.apiClient);
|
|
694
816
|
case 'generate_image':
|
|
695
817
|
return handleGenerateImage(wrapToolArgs(args), this.openai);
|
|
818
|
+
case 'generate_image_dedicated':
|
|
819
|
+
return handleGenerateImageDedicated(wrapToolArgs(args), this.apiClient);
|
|
696
820
|
case 'generate_audio':
|
|
697
821
|
return handleGenerateAudio(wrapToolArgs(args), this.openai);
|
|
822
|
+
case 'text_to_speech':
|
|
823
|
+
return handleTextToSpeech(wrapToolArgs(args), this.apiClient);
|
|
824
|
+
case 'speech_to_text':
|
|
825
|
+
return handleSpeechToText(wrapToolArgs(args), this.apiClient);
|
|
698
826
|
case 'generate_video':
|
|
699
827
|
return handleGenerateVideo(wrapToolArgs(args), this.apiClient, buildProgressHook(this.server, extractProgressToken(request)));
|
|
700
828
|
case 'generate_video_from_image':
|
package/dist/version.d.ts
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
* Bumped in lockstep with package.json / server.json / smithery.yaml /
|
|
7
7
|
* scripts/build-manifest.mjs during release prep.
|
|
8
8
|
*/
|
|
9
|
-
export declare const SERVER_VERSION = "4.
|
|
9
|
+
export declare const SERVER_VERSION = "4.6.0";
|
|
10
10
|
/**
|
|
11
11
|
* MCP protocol version our SDK speaks. Hardcoded to match the version
|
|
12
12
|
* bundled with `@modelcontextprotocol/sdk`; update when upgrading the
|
package/dist/version.js
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
* Bumped in lockstep with package.json / server.json / smithery.yaml /
|
|
7
7
|
* scripts/build-manifest.mjs during release prep.
|
|
8
8
|
*/
|
|
9
|
-
export const SERVER_VERSION = '4.
|
|
9
|
+
export const SERVER_VERSION = '4.6.0';
|
|
10
10
|
/**
|
|
11
11
|
* MCP protocol version our SDK speaks. Hardcoded to match the version
|
|
12
12
|
* bundled with `@modelcontextprotocol/sdk`; update when upgrading the
|
package/package.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@stabgan/openrouter-mcp-multimodal",
|
|
3
|
-
"version": "4.
|
|
3
|
+
"version": "4.6.0",
|
|
4
4
|
"mcpName": "io.github.stabgan/openrouter-multimodal",
|
|
5
|
-
"description": "MCP server for OpenRouter
|
|
5
|
+
"description": "MCP server for OpenRouter — chat with 300+ LLMs, analyze images/audio/video, generate images (dedicated API), TTS/STT, video generation (Veo 3.1 / Seedance / Wan), async completions, response caching",
|
|
6
6
|
"type": "module",
|
|
7
7
|
"main": "dist/index.js",
|
|
8
8
|
"bin": {
|
|
@@ -19,8 +19,18 @@
|
|
|
19
19
|
"lint": "eslint src",
|
|
20
20
|
"format": "prettier --write \"src/**/*.ts\" *.json \"*.md\"",
|
|
21
21
|
"format:check": "prettier --check \"src/**/*.ts\"",
|
|
22
|
-
"test": "
|
|
23
|
-
"test:
|
|
22
|
+
"test": "vitest run",
|
|
23
|
+
"test:regression": "vitest run --config vitest.regression.config.ts",
|
|
24
|
+
"test:integration": "vitest run --config vitest.integration.config.ts",
|
|
25
|
+
"test:e2e": "npm run build && node scripts/live-e2e.mjs",
|
|
26
|
+
"test:smoke:npm": "npm run build && npm pack --quiet && node scripts/smoke-npm-mcp.mjs",
|
|
27
|
+
"test:smoke:docker": "node scripts/smoke-docker-mcp.mjs",
|
|
28
|
+
"test:smoke:uvx": "node scripts/smoke-uvx-mcp.mjs",
|
|
29
|
+
"test:smoke:uvx:local": "npm run build && npm pack --quiet && MCP_UVX_LOCAL=1 node scripts/smoke-uvx-mcp.mjs",
|
|
30
|
+
"test:smoke:uvx:git": "MCP_UVX_FROM_GIT=1 node scripts/smoke-uvx-mcp.mjs",
|
|
31
|
+
"test:smoke": "npm run test:smoke:npm && npm run test:smoke:docker && npm run test:smoke:uvx:local",
|
|
32
|
+
"test:all": "npm test && npm run test:regression && npm run test:integration",
|
|
33
|
+
"ci": "npm run lint && npm run format:check && npm run build && npm run test:all"
|
|
24
34
|
},
|
|
25
35
|
"keywords": [
|
|
26
36
|
"mcp",
|
|
@@ -44,23 +54,25 @@
|
|
|
44
54
|
"homepage": "https://github.com/stabgan/openrouter-mcp-multimodal#readme",
|
|
45
55
|
"license": "Apache-2.0",
|
|
46
56
|
"engines": {
|
|
47
|
-
"node": ">=
|
|
57
|
+
"node": ">=20.0.0"
|
|
48
58
|
},
|
|
49
59
|
"dependencies": {
|
|
50
|
-
"@modelcontextprotocol/sdk": "^1.
|
|
51
|
-
"dotenv": "^
|
|
52
|
-
"openai": "^4.
|
|
53
|
-
"sharp": "^0.
|
|
60
|
+
"@modelcontextprotocol/sdk": "^1.29.0",
|
|
61
|
+
"dotenv": "^17.4.2",
|
|
62
|
+
"openai": "^4.104.0",
|
|
63
|
+
"sharp": "^0.35.3"
|
|
54
64
|
},
|
|
55
65
|
"devDependencies": {
|
|
56
66
|
"@eslint/js": "^9.39.2",
|
|
57
|
-
"@types/node": "^22.
|
|
67
|
+
"@types/node": "^22.20.0",
|
|
58
68
|
"eslint": "^9.39.2",
|
|
59
69
|
"eslint-config-prettier": "^10.1.8",
|
|
60
|
-
"
|
|
61
|
-
"
|
|
62
|
-
"
|
|
63
|
-
"
|
|
64
|
-
"
|
|
70
|
+
"form-data": "^4.0.6",
|
|
71
|
+
"js-yaml": "^4.3.0",
|
|
72
|
+
"prettier": "^3.9.4",
|
|
73
|
+
"shx": "^0.4.0",
|
|
74
|
+
"typescript": "^5.9.3",
|
|
75
|
+
"typescript-eslint": "^8.62.1",
|
|
76
|
+
"vitest": "^4.1.9"
|
|
65
77
|
}
|
|
66
78
|
}
|