@stabgan/openrouter-mcp-multimodal 4.7.0 → 5.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +95 -41
  2. package/dist/errors.d.ts +5 -20
  3. package/dist/errors.js +1 -10
  4. package/dist/index.js +8 -2
  5. package/dist/logger.js +54 -24
  6. package/dist/model-cache.d.ts +13 -0
  7. package/dist/model-cache.js +62 -8
  8. package/dist/openrouter-api.d.ts +14 -15
  9. package/dist/openrouter-api.js +68 -22
  10. package/dist/tool-definitions.d.ts +24 -0
  11. package/dist/tool-definitions.js +280 -177
  12. package/dist/tool-descriptions.d.ts +0 -4
  13. package/dist/tool-descriptions.js +30 -21
  14. package/dist/tool-handlers/analyze-audio.js +4 -1
  15. package/dist/tool-handlers/analyze-image.js +11 -6
  16. package/dist/tool-handlers/analyze-video.js +9 -5
  17. package/dist/tool-handlers/async-chat.d.ts +17 -0
  18. package/dist/tool-handlers/async-chat.js +104 -30
  19. package/dist/tool-handlers/audio-utils.d.ts +19 -4
  20. package/dist/tool-handlers/audio-utils.js +170 -16
  21. package/dist/tool-handlers/cache.d.ts +3 -3
  22. package/dist/tool-handlers/cache.js +56 -4
  23. package/dist/tool-handlers/chat-completion.js +16 -7
  24. package/dist/tool-handlers/chat-request.d.ts +4 -1
  25. package/dist/tool-handlers/chat-request.js +29 -1
  26. package/dist/tool-handlers/completion-utils.d.ts +5 -11
  27. package/dist/tool-handlers/completion-utils.js +76 -47
  28. package/dist/tool-handlers/fetch-utils.d.ts +14 -0
  29. package/dist/tool-handlers/fetch-utils.js +329 -77
  30. package/dist/tool-handlers/generate-audio.d.ts +4 -15
  31. package/dist/tool-handlers/generate-audio.js +21 -53
  32. package/dist/tool-handlers/generate-image-dedicated.d.ts +1 -1
  33. package/dist/tool-handlers/generate-image-dedicated.js +50 -31
  34. package/dist/tool-handlers/generate-image.d.ts +1 -1
  35. package/dist/tool-handlers/generate-image.js +19 -22
  36. package/dist/tool-handlers/generate-video.d.ts +4 -3
  37. package/dist/tool-handlers/generate-video.js +42 -18
  38. package/dist/tool-handlers/get-model-info.js +1 -1
  39. package/dist/tool-handlers/health-check.js +39 -15
  40. package/dist/tool-handlers/image-utils.js +2 -2
  41. package/dist/tool-handlers/openrouter-errors.d.ts +2 -0
  42. package/dist/tool-handlers/openrouter-errors.js +138 -31
  43. package/dist/tool-handlers/path-safety.js +49 -17
  44. package/dist/tool-handlers/path-utils.d.ts +2 -0
  45. package/dist/tool-handlers/path-utils.js +13 -0
  46. package/dist/tool-handlers/provider-routing.d.ts +2 -0
  47. package/dist/tool-handlers/provider-routing.js +11 -1
  48. package/dist/tool-handlers/rerank.d.ts +1 -4
  49. package/dist/tool-handlers/rerank.js +44 -15
  50. package/dist/tool-handlers/search-models.js +3 -3
  51. package/dist/tool-handlers/speech-to-text.d.ts +1 -0
  52. package/dist/tool-handlers/speech-to-text.js +23 -56
  53. package/dist/tool-handlers/text-to-speech.d.ts +1 -1
  54. package/dist/tool-handlers/text-to-speech.js +33 -20
  55. package/dist/tool-handlers/tool-result-payload.js +11 -9
  56. package/dist/tool-handlers/validate-model.js +1 -1
  57. package/dist/tool-handlers.d.ts +9 -0
  58. package/dist/tool-handlers.js +17 -5
  59. package/dist/tts-defaults.d.ts +4 -0
  60. package/dist/tts-defaults.js +4 -0
  61. package/dist/version.d.ts +3 -2
  62. package/dist/version.js +4 -2
  63. package/package.json +11 -13
@@ -1,17 +1,19 @@
1
- /** Dedicated POST /api/v1/audio/speech — OpenAI, Gemini Flash TTS, Voxtral. */
2
- import { promises as fs } from 'node:fs';
1
+ /** Dedicated POST /api/v1/audio/speech. */
3
2
  import { extname } from 'node:path';
3
+ import { TTS_RESPONSE_FORMATS } from '../tool-definitions.js';
4
+ import { DEFAULT_TTS_MODEL, DEFAULT_TTS_RESPONSE_FORMAT, DEFAULT_TTS_VOICE, } from '../tts-defaults.js';
4
5
  import { resolveOptionalOutputPath, isToolErrorResult } from './path-safety.js';
5
6
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
6
7
  import { SERVER_VERSION } from '../version.js';
7
8
  import { logger } from '../logger.js';
8
9
  import { classifyUpstreamError } from './openrouter-errors.js';
9
10
  import { buildBinaryToolResult } from './tool-result-payload.js';
10
- import { replaceExtension } from './path-utils.js';
11
- import { buildCacheHeaders } from './cache.js';
12
- const DEFAULT_MODEL = 'openai/gpt-4o-mini-tts-2025-12-15';
13
- const DEFAULT_VOICE = 'alloy';
14
- const VALID_FORMATS = new Set(['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm']);
11
+ import { replaceExtension, writeOutputFile } from './path-utils.js';
12
+ import { buildCacheHeaders, validateCacheOptions } from './cache.js';
13
+ import { detectAudioFormat } from './audio-utils.js';
14
+ const MIN_SPEED = 0.25;
15
+ const MAX_SPEED = 4.0;
16
+ const VALID_FORMATS = new Set(TTS_RESPONSE_FORMATS);
15
17
  export async function handleTextToSpeech(request, apiClient) {
16
18
  const args = request.params.arguments ?? {};
17
19
  const { input, model, voice, response_format, speed, instructions, save_path, cache, cache_ttl, cache_clear, } = args;
@@ -21,10 +23,19 @@ export async function handleTextToSpeech(request, apiClient) {
21
23
  if (response_format && !VALID_FORMATS.has(response_format)) {
22
24
  return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_FORMATS].join(', ')}.`);
23
25
  }
26
+ if (typeof speed === 'number' && (speed < MIN_SPEED || speed > MAX_SPEED)) {
27
+ return toolError(ErrorCode.INVALID_INPUT, `speed must be between ${MIN_SPEED} and ${MAX_SPEED} (inclusive).`);
28
+ }
29
+ const cacheError = validateCacheOptions({ cache, cache_ttl, cache_clear });
30
+ if (cacheError)
31
+ return cacheError;
32
+ const effectiveModel = model?.trim() || DEFAULT_TTS_MODEL;
33
+ const effectiveVoice = voice?.trim() || (effectiveModel === DEFAULT_TTS_MODEL ? DEFAULT_TTS_VOICE : undefined);
34
+ const effectiveResponseFormat = response_format || DEFAULT_TTS_RESPONSE_FORMAT;
24
35
  logger.audit('text_to_speech.start', {
25
- model: model || DEFAULT_MODEL,
26
- voice: voice || DEFAULT_VOICE,
27
- response_format: response_format || 'mp3',
36
+ model: effectiveModel,
37
+ voice: effectiveVoice || 'provider default',
38
+ response_format: effectiveResponseFormat,
28
39
  input_preview: input.slice(0, 80),
29
40
  save_path: save_path ? 'provided' : 'none',
30
41
  });
@@ -33,13 +44,13 @@ export async function handleTextToSpeech(request, apiClient) {
33
44
  return savePathResult;
34
45
  const safeSavePath = savePathResult.path;
35
46
  const body = {
36
- model: model || DEFAULT_MODEL,
47
+ model: effectiveModel,
37
48
  input,
38
- voice: voice || DEFAULT_VOICE,
49
+ response_format: effectiveResponseFormat,
39
50
  };
40
- if (response_format)
41
- body.response_format = response_format;
42
- if (typeof speed === 'number' && speed > 0)
51
+ if (effectiveVoice)
52
+ body.voice = effectiveVoice;
53
+ if (typeof speed === 'number')
43
54
  body.speed = speed;
44
55
  if (instructions)
45
56
  body.instructions = instructions;
@@ -52,20 +63,22 @@ export async function handleTextToSpeech(request, apiClient) {
52
63
  return classifyUpstreamError(err, 'text_to_speech');
53
64
  }
54
65
  const { buffer, contentType } = result;
55
- const mimeType = contentType.split(';')[0]?.trim() || 'audio/mpeg';
56
- const ext = response_format || 'mp3';
66
+ const detected = detectAudioFormat(buffer);
67
+ const mimeType = detected.mimeType || contentType.split(';')[0]?.trim() || 'audio/mpeg';
68
+ const ext = detected.ext;
57
69
  const baseMeta = {
58
70
  server_version: SERVER_VERSION,
59
- model: model || DEFAULT_MODEL,
71
+ model: effectiveModel,
60
72
  mime: mimeType,
61
73
  size_bytes: buffer.length,
62
- voice: voice || DEFAULT_VOICE,
63
74
  };
75
+ if (effectiveVoice)
76
+ baseMeta.voice = effectiveVoice;
64
77
  if (safeSavePath) {
65
78
  const currentExt = extname(safeSavePath).toLowerCase().slice(1);
66
79
  const actualPath = currentExt === ext ? safeSavePath : replaceExtension(safeSavePath, ext);
67
80
  try {
68
- await fs.writeFile(actualPath, buffer);
81
+ await writeOutputFile(actualPath, buffer);
69
82
  }
70
83
  catch (err) {
71
84
  return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
@@ -1,10 +1,3 @@
1
- /**
2
- * Canonical MCP tool-result policy for binary artifacts:
3
- * - saved to disk -> text pointer only (never duplicate inline media)
4
- * - not saved -> inline only when under byte ceiling, else text + save_path hint
5
- * - video uses MCP `resource` blocks (spec has no `video` content type)
6
- */
7
- import { readEnvInt } from './fetch-utils.js';
8
1
  const DEFAULT_INLINE_MAX_BYTES = 1024 * 1024;
9
2
  const DEFAULT_VIDEO_INLINE_MAX_BYTES = 10 * 1024 * 1024;
10
3
  const INLINE_VIDEO_URI = 'inline://openrouter-mcp-multimodal/video';
@@ -18,9 +11,18 @@ const KIND_DEFAULT_BYTES = {
18
11
  audio: DEFAULT_INLINE_MAX_BYTES,
19
12
  video: DEFAULT_VIDEO_INLINE_MAX_BYTES,
20
13
  };
14
+ function readEnvInlineMaxBytes(name, fallback) {
15
+ const raw = process.env[name];
16
+ if (raw === undefined || raw === '')
17
+ return fallback;
18
+ if (!/^\d+$/.test(raw))
19
+ return fallback;
20
+ const n = Number(raw);
21
+ return Number.isFinite(n) && n >= 0 ? n : fallback;
22
+ }
21
23
  export function getMaxInlineBytes(kind) {
22
- const globalFallback = readEnvInt('OPENROUTER_INLINE_MAX_BYTES', KIND_DEFAULT_BYTES[kind], 4096);
23
- return readEnvInt(KIND_ENV_KEYS[kind], globalFallback, 4096);
24
+ const globalFallback = readEnvInlineMaxBytes('OPENROUTER_INLINE_MAX_BYTES', KIND_DEFAULT_BYTES[kind]);
25
+ return readEnvInlineMaxBytes(KIND_ENV_KEYS[kind], globalFallback);
24
26
  }
25
27
  function kindLabel(kind) {
26
28
  switch (kind) {
@@ -17,6 +17,6 @@ export async function handleValidateModel(request, modelCache, apiClient) {
17
17
  if (!modelCache.isValid()) {
18
18
  return toolError(ErrorCode.INTERNAL, 'No model data available.');
19
19
  }
20
- const valid = modelCache.has(model);
20
+ const valid = modelCache.catalogHas(model);
21
21
  return buildStructuredResult({ valid, model });
22
22
  }
@@ -1,4 +1,12 @@
1
1
  import { Server } from '@modelcontextprotocol/sdk/server/index.js';
2
+ type McpProgressHook = (update: {
3
+ status: string;
4
+ progress?: number;
5
+ attempt: number;
6
+ video_id: string;
7
+ }) => void;
8
+ /** MCP `notifications/progress` hook. */
9
+ export declare function buildProgressHook(server: Server, progressToken: string | number | undefined): McpProgressHook | undefined;
2
10
  export declare class ToolHandlers {
3
11
  private openai;
4
12
  private modelCache;
@@ -8,3 +16,4 @@ export declare class ToolHandlers {
8
16
  constructor(server: Server, apiKey: string, defaultModel?: string);
9
17
  private register;
10
18
  }
19
+ export {};
@@ -20,25 +20,31 @@ import { handleSpeechToText } from './tool-handlers/speech-to-text.js';
20
20
  import { handleStartChatCompletion, handleGetChatCompletionStatus, } from './tool-handlers/async-chat.js';
21
21
  import { TOOL_DEFINITIONS } from './tool-definitions.js';
22
22
  import { TOOL_ICONS } from './tool-icons.js';
23
+ import { ErrorCode, toolErrorFrom } from './errors.js';
24
+ import { logger } from './logger.js';
23
25
  function wrapToolArgs(a) {
24
26
  return { params: { arguments: a ?? {} } };
25
27
  }
26
- function buildProgressHook(server, progressToken) {
28
+ /** MCP `notifications/progress` hook. */
29
+ export function buildProgressHook(server, progressToken) {
27
30
  if (progressToken === undefined)
28
31
  return undefined;
29
- // MCP progress must monotonically increase; upstream values can drop or be omitted.
30
32
  let lastSent = -1;
31
33
  return ({ status, progress, attempt, video_id }) => {
32
34
  const candidate = typeof progress === 'number' ? Math.max(attempt, progress) : attempt;
33
35
  const next = Math.max(lastSent + 1, candidate);
34
36
  lastSent = next;
35
- void server.notification({
37
+ void server
38
+ .notification({
36
39
  method: 'notifications/progress',
37
40
  params: {
38
41
  progressToken,
39
42
  progress: next,
40
43
  message: `video ${video_id} — ${status}${typeof progress === 'number' ? ` (${progress}%)` : ''}`,
41
44
  },
45
+ })
46
+ .catch((err) => {
47
+ logger.warn('progress notification failed', { err: String(err) });
42
48
  });
43
49
  };
44
50
  }
@@ -68,7 +74,6 @@ export class ToolHandlers {
68
74
  }));
69
75
  server.setRequestHandler(CallToolRequestSchema, async (request) => {
70
76
  const { name, arguments: args } = request.params;
71
- // Handlers return extra _meta keys not in the SDK type.
72
77
  const dispatch = async () => {
73
78
  switch (name) {
74
79
  case 'chat_completion':
@@ -113,7 +118,14 @@ export class ToolHandlers {
113
118
  throw new McpError(McpErrorCode.MethodNotFound, `Unknown tool: ${name}`);
114
119
  }
115
120
  };
116
- return (await dispatch());
121
+ try {
122
+ return (await dispatch());
123
+ }
124
+ catch (err) {
125
+ if (err instanceof McpError)
126
+ throw err;
127
+ return toolErrorFrom(ErrorCode.INTERNAL, err);
128
+ }
117
129
  });
118
130
  }
119
131
  }
@@ -0,0 +1,4 @@
1
+ /** Current free OpenRouter TTS defaults; model and voice can be overridden per call. */
2
+ export declare const DEFAULT_TTS_MODEL = "deepgram/flux-tts:free";
3
+ export declare const DEFAULT_TTS_VOICE = "flux-alexis-en";
4
+ export declare const DEFAULT_TTS_RESPONSE_FORMAT = "mp3";
@@ -0,0 +1,4 @@
1
+ /** Current free OpenRouter TTS defaults; model and voice can be overridden per call. */
2
+ export const DEFAULT_TTS_MODEL = 'deepgram/flux-tts:free';
3
+ export const DEFAULT_TTS_VOICE = 'flux-alexis-en';
4
+ export const DEFAULT_TTS_RESPONSE_FORMAT = 'mp3';
package/dist/version.d.ts CHANGED
@@ -1,2 +1,3 @@
1
- export declare const SERVER_VERSION = "4.7.0";
2
- export declare const MCP_PROTOCOL_VERSION = "2025-06-18";
1
+ export declare const SERVER_VERSION = "5.0.0";
2
+ /** Advertised MCP protocol version — derived from the installed SDK. */
3
+ export declare const MCP_PROTOCOL_VERSION = "2025-11-25";
package/dist/version.js CHANGED
@@ -1,2 +1,4 @@
1
- export const SERVER_VERSION = '4.7.0';
2
- export const MCP_PROTOCOL_VERSION = '2025-06-18';
1
+ import { LATEST_PROTOCOL_VERSION } from '@modelcontextprotocol/sdk/types.js';
2
+ export const SERVER_VERSION = '5.0.0';
3
+ /** Advertised MCP protocol version — derived from the installed SDK. */
4
+ export const MCP_PROTOCOL_VERSION = LATEST_PROTOCOL_VERSION;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@stabgan/openrouter-mcp-multimodal",
3
- "version": "4.7.0",
3
+ "version": "5.0.0",
4
4
  "mcpName": "io.github.stabgan/openrouter-multimodal",
5
5
  "description": "MCP server for OpenRouter — chat with 300+ LLMs, analyze images/audio/video, generate images (dedicated API), TTS/STT, video generation (Veo 3.1 / Seedance / Wan), async completions, response caching",
6
6
  "type": "module",
@@ -55,25 +55,23 @@
55
55
  "homepage": "https://github.com/stabgan/openrouter-mcp-multimodal#readme",
56
56
  "license": "Apache-2.0",
57
57
  "engines": {
58
- "node": ">=20.0.0"
58
+ "node": ">=22.0.0"
59
59
  },
60
60
  "dependencies": {
61
- "@modelcontextprotocol/sdk": "^1.29.0",
61
+ "@modelcontextprotocol/sdk": "^1.30.0",
62
62
  "dotenv": "^17.4.2",
63
- "openai": "^4.104.0",
64
- "sharp": "^0.35.3"
63
+ "openai": "^7.9.0",
64
+ "sharp": "^0.35.4"
65
65
  },
66
66
  "devDependencies": {
67
- "@eslint/js": "^9.39.2",
68
- "@types/node": "^22.20.0",
69
- "eslint": "^9.39.2",
67
+ "@eslint/js": "^10.0.1",
68
+ "@types/node": "^22.20.1",
69
+ "eslint": "^10.9.1",
70
70
  "eslint-config-prettier": "^10.1.8",
71
- "form-data": "^4.0.6",
72
- "js-yaml": "^4.3.0",
73
- "prettier": "^3.9.4",
71
+ "prettier": "^3.9.6",
74
72
  "shx": "^0.4.0",
75
73
  "typescript": "^5.9.3",
76
- "typescript-eslint": "^8.62.1",
77
- "vitest": "^4.1.9"
74
+ "typescript-eslint": "^8.69.0",
75
+ "vitest": "^4.1.11"
78
76
  }
79
77
  }