@stabgan/openrouter-mcp-multimodal 4.5.3 → 4.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,6 +14,10 @@ import { handleAnalyzeVideo } from './tool-handlers/analyze-video.js';
14
14
  import { handleGenerateVideo, handleGetVideoStatus, handleGenerateVideoFromImage, } from './tool-handlers/generate-video.js';
15
15
  import { handleRerankDocuments } from './tool-handlers/rerank.js';
16
16
  import { handleHealthCheck } from './tool-handlers/health-check.js';
17
+ import { handleGenerateImageDedicated, } from './tool-handlers/generate-image-dedicated.js';
18
+ import { handleTextToSpeech } from './tool-handlers/text-to-speech.js';
19
+ import { handleSpeechToText } from './tool-handlers/speech-to-text.js';
20
+ import { handleStartChatCompletion, handleGetChatCompletionStatus, } from './tool-handlers/async-chat.js';
17
21
  import { TOOL_DESCRIPTIONS } from './tool-descriptions.js';
18
22
  function wrapToolArgs(a) {
19
23
  return { params: { arguments: a ?? {} } };
@@ -88,8 +92,9 @@ export class ToolHandlers {
88
92
  model: {
89
93
  type: 'string',
90
94
  description: 'Model ID (optional, uses default). Append `:nitro` for the fastest variant, ' +
91
- '`:floor` for the cheapest, or `:exacto` for the best tool-calling accuracy. ' +
92
- 'Example: `openai/gpt-4o:nitro`.',
95
+ '`:floor` for the cheapest, `:free` for the free tier, `:online` for web search, ' +
96
+ 'or `:exacto` for the best tool-calling accuracy. ' +
97
+ 'Example: `openai/gpt-4o:nitro`. Or pass `online: true` for programmatic web search control.',
93
98
  },
94
99
  messages: {
95
100
  type: 'array',
@@ -166,6 +171,63 @@ export class ToolHandlers {
166
171
  required: ['messages'],
167
172
  },
168
173
  },
174
+ {
175
+ name: 'start_chat_completion',
176
+ description: TOOL_DESCRIPTIONS.start_chat_completion,
177
+ annotations: {
178
+ title: 'Start async chat completion',
179
+ readOnlyHint: false,
180
+ destructiveHint: false,
181
+ idempotentHint: false,
182
+ openWorldHint: true,
183
+ },
184
+ inputSchema: {
185
+ type: 'object',
186
+ properties: {
187
+ model: { type: 'string', description: 'Model ID (same as chat_completion).' },
188
+ messages: {
189
+ type: 'array',
190
+ minItems: 1,
191
+ items: {
192
+ type: 'object',
193
+ properties: {
194
+ role: { type: 'string', enum: ['system', 'user', 'assistant'] },
195
+ content: { oneOf: [{ type: 'string' }, { type: 'array', items: { type: 'object' } }] },
196
+ },
197
+ required: ['role', 'content'],
198
+ },
199
+ },
200
+ temperature: { type: 'number', minimum: 0, maximum: 2 },
201
+ max_tokens: { type: 'number', minimum: 1 },
202
+ provider: { type: 'object' },
203
+ include_reasoning: { type: 'boolean' },
204
+ online: { type: 'boolean' },
205
+ web_max_results: { type: 'number', minimum: 1 },
206
+ cache: { type: 'boolean' },
207
+ cache_ttl: { type: 'string' },
208
+ cache_clear: { type: 'boolean' },
209
+ },
210
+ required: ['messages'],
211
+ },
212
+ },
213
+ {
214
+ name: 'get_chat_completion_status',
215
+ description: TOOL_DESCRIPTIONS.get_chat_completion_status,
216
+ annotations: {
217
+ title: 'Get async chat completion status',
218
+ readOnlyHint: true,
219
+ destructiveHint: false,
220
+ idempotentHint: true,
221
+ openWorldHint: false,
222
+ },
223
+ inputSchema: {
224
+ type: 'object',
225
+ properties: {
226
+ job_id: { type: 'string', description: 'The job_id returned by start_chat_completion.' },
227
+ },
228
+ required: ['job_id'],
229
+ },
230
+ },
169
231
  {
170
232
  name: 'analyze_image',
171
233
  description: TOOL_DESCRIPTIONS.analyze_image,
@@ -394,6 +456,69 @@ export class ToolHandlers {
394
456
  required: ['prompt'],
395
457
  },
396
458
  },
459
+ {
460
+ name: 'generate_image_dedicated',
461
+ description: TOOL_DESCRIPTIONS.generate_image_dedicated,
462
+ annotations: {
463
+ title: 'Generate image (dedicated API)',
464
+ readOnlyHint: false,
465
+ destructiveHint: false,
466
+ idempotentHint: false,
467
+ openWorldHint: true,
468
+ },
469
+ inputSchema: {
470
+ type: 'object',
471
+ properties: {
472
+ prompt: {
473
+ type: 'string',
474
+ description: 'Text prompt describing the image to generate.',
475
+ },
476
+ model: {
477
+ type: 'string',
478
+ description: 'Image model ID. Default: google/gemini-2.5-flash-image. Browse available at https://openrouter.ai/collections/image-models',
479
+ },
480
+ resolution: {
481
+ type: 'string',
482
+ enum: ['512', '0.5K', '1K', '2K', '4K'],
483
+ description: 'Normalized resolution tier. Provider maps to closest supported size.',
484
+ },
485
+ aspect_ratio: {
486
+ type: 'string',
487
+ description: 'Aspect ratio (e.g. "1:1", "16:9", "9:16", "4:3", "21:9").',
488
+ },
489
+ quality: {
490
+ type: 'string',
491
+ enum: ['auto', 'low', 'medium', 'high'],
492
+ description: 'Image quality level.',
493
+ },
494
+ output_format: {
495
+ type: 'string',
496
+ enum: ['png', 'jpeg', 'webp', 'svg'],
497
+ description: 'Output image format.',
498
+ },
499
+ n: {
500
+ type: 'number',
501
+ minimum: 1,
502
+ maximum: 10,
503
+ description: 'Number of images to generate (model-dependent, default 1).',
504
+ },
505
+ input_references: {
506
+ type: 'array',
507
+ items: { type: 'string' },
508
+ description: 'Reference images for image-to-image workflows. Each entry: local path, http(s) URL, or data URL.',
509
+ },
510
+ save_path: { type: 'string', description: 'Save generated image to this path.' },
511
+ provider: {
512
+ type: 'object',
513
+ description: 'Provider routing overrides (order, sort, allow_fallbacks, etc.).',
514
+ },
515
+ cache: { type: 'boolean' },
516
+ cache_ttl: { type: 'string' },
517
+ cache_clear: { type: 'boolean' },
518
+ },
519
+ required: ['prompt'],
520
+ },
521
+ },
397
522
  {
398
523
  name: 'generate_audio',
399
524
  description: TOOL_DESCRIPTIONS.generate_audio,
@@ -416,6 +541,97 @@ export class ToolHandlers {
416
541
  required: ['prompt'],
417
542
  },
418
543
  },
544
+ {
545
+ name: 'text_to_speech',
546
+ description: TOOL_DESCRIPTIONS.text_to_speech,
547
+ annotations: {
548
+ title: 'Text to speech (dedicated API)',
549
+ readOnlyHint: false,
550
+ destructiveHint: false,
551
+ idempotentHint: false,
552
+ openWorldHint: true,
553
+ },
554
+ inputSchema: {
555
+ type: 'object',
556
+ properties: {
557
+ input: {
558
+ type: 'string',
559
+ description: 'Text to convert to speech.',
560
+ },
561
+ model: {
562
+ type: 'string',
563
+ description: 'TTS model. Default: openai/gpt-4o-mini-tts-2025-12-15. Also: google/gemini-flash-tts, mistral/voxtral-mini-tts.',
564
+ },
565
+ voice: {
566
+ type: 'string',
567
+ description: 'Voice ID (model-specific). Default: alloy. OpenAI voices: alloy, echo, fable, onyx, nova, shimmer.',
568
+ },
569
+ response_format: {
570
+ type: 'string',
571
+ enum: ['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm'],
572
+ description: 'Output audio format. Default: mp3.',
573
+ },
574
+ speed: {
575
+ type: 'number',
576
+ minimum: 0.25,
577
+ maximum: 4.0,
578
+ description: 'Speed of speech (0.25 to 4.0). Default: 1.0.',
579
+ },
580
+ instructions: {
581
+ type: 'string',
582
+ description: 'Tone/style instructions (e.g. "speak in a warm, friendly tone"). OpenAI models only.',
583
+ },
584
+ save_path: { type: 'string', description: 'Save audio to this path.' },
585
+ cache: { type: 'boolean' },
586
+ cache_ttl: { type: 'string' },
587
+ cache_clear: { type: 'boolean' },
588
+ },
589
+ required: ['input'],
590
+ },
591
+ },
592
+ {
593
+ name: 'speech_to_text',
594
+ description: TOOL_DESCRIPTIONS.speech_to_text,
595
+ annotations: {
596
+ title: 'Speech to text (dedicated API)',
597
+ readOnlyHint: true,
598
+ destructiveHint: false,
599
+ idempotentHint: false,
600
+ openWorldHint: true,
601
+ },
602
+ inputSchema: {
603
+ type: 'object',
604
+ properties: {
605
+ audio_path: {
606
+ type: 'string',
607
+ description: 'Audio file: local path (sandboxed), http(s) URL, or base64 data URL. Formats: mp3, wav, flac, ogg, webm, mp4.',
608
+ },
609
+ model: {
610
+ type: 'string',
611
+ description: 'STT model. Default: openai/whisper-1. Also: openai/gpt-4o-transcribe, openai/gpt-4o-mini-transcribe.',
612
+ },
613
+ language: {
614
+ type: 'string',
615
+ description: 'ISO-639-1 language code (e.g. "en", "es", "fr"). Improves accuracy.',
616
+ },
617
+ response_format: {
618
+ type: 'string',
619
+ enum: ['json', 'text', 'srt', 'verbose_json', 'vtt'],
620
+ description: 'Output format for transcription. Default: json.',
621
+ },
622
+ temperature: {
623
+ type: 'number',
624
+ minimum: 0,
625
+ maximum: 1,
626
+ description: 'Sampling temperature for transcription (0-1).',
627
+ },
628
+ cache: { type: 'boolean' },
629
+ cache_ttl: { type: 'string' },
630
+ cache_clear: { type: 'boolean' },
631
+ },
632
+ required: ['audio_path'],
633
+ },
634
+ },
419
635
  {
420
636
  name: 'generate_video',
421
637
  description: TOOL_DESCRIPTIONS.generate_video,
@@ -581,6 +797,10 @@ export class ToolHandlers {
581
797
  switch (name) {
582
798
  case 'chat_completion':
583
799
  return handleChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
800
+ case 'start_chat_completion':
801
+ return handleStartChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
802
+ case 'get_chat_completion_status':
803
+ return handleGetChatCompletionStatus(wrapToolArgs(args));
584
804
  case 'analyze_image':
585
805
  return handleAnalyzeImage(wrapToolArgs(args), this.openai, this.defaultModel);
586
806
  case 'analyze_audio':
@@ -595,8 +815,14 @@ export class ToolHandlers {
595
815
  return handleValidateModel(wrapToolArgs(args), this.modelCache, this.apiClient);
596
816
  case 'generate_image':
597
817
  return handleGenerateImage(wrapToolArgs(args), this.openai);
818
+ case 'generate_image_dedicated':
819
+ return handleGenerateImageDedicated(wrapToolArgs(args), this.apiClient);
598
820
  case 'generate_audio':
599
821
  return handleGenerateAudio(wrapToolArgs(args), this.openai);
822
+ case 'text_to_speech':
823
+ return handleTextToSpeech(wrapToolArgs(args), this.apiClient);
824
+ case 'speech_to_text':
825
+ return handleSpeechToText(wrapToolArgs(args), this.apiClient);
600
826
  case 'generate_video':
601
827
  return handleGenerateVideo(wrapToolArgs(args), this.apiClient, buildProgressHook(this.server, extractProgressToken(request)));
602
828
  case 'generate_video_from_image':
package/dist/version.d.ts CHANGED
@@ -6,7 +6,7 @@
6
6
  * Bumped in lockstep with package.json / server.json / smithery.yaml /
7
7
  * scripts/build-manifest.mjs during release prep.
8
8
  */
9
- export declare const SERVER_VERSION = "4.5.3";
9
+ export declare const SERVER_VERSION = "4.6.0";
10
10
  /**
11
11
  * MCP protocol version our SDK speaks. Hardcoded to match the version
12
12
  * bundled with `@modelcontextprotocol/sdk`; update when upgrading the
package/dist/version.js CHANGED
@@ -6,7 +6,7 @@
6
6
  * Bumped in lockstep with package.json / server.json / smithery.yaml /
7
7
  * scripts/build-manifest.mjs during release prep.
8
8
  */
9
- export const SERVER_VERSION = '4.5.3';
9
+ export const SERVER_VERSION = '4.6.0';
10
10
  /**
11
11
  * MCP protocol version our SDK speaks. Hardcoded to match the version
12
12
  * bundled with `@modelcontextprotocol/sdk`; update when upgrading the
package/package.json CHANGED
@@ -1,8 +1,8 @@
1
1
  {
2
2
  "name": "@stabgan/openrouter-mcp-multimodal",
3
- "version": "4.5.3",
3
+ "version": "4.6.0",
4
4
  "mcpName": "io.github.stabgan/openrouter-multimodal",
5
- "description": "MCP server for OpenRouter with text chat, image analysis + generation, audio analysis + generation, video analysis, and video generation (Veo 3.1 / Sora 2 Pro / Seedance / Wan)",
5
+ "description": "MCP server for OpenRouter — chat with 300+ LLMs, analyze images/audio/video, generate images (dedicated API), TTS/STT, video generation (Veo 3.1 / Seedance / Wan), async completions, response caching",
6
6
  "type": "module",
7
7
  "main": "dist/index.js",
8
8
  "bin": {