@stabgan/openrouter-mcp-multimodal 4.0.0 → 4.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +40 -7
  2. package/dist/errors.d.ts +26 -8
  3. package/dist/errors.js +17 -12
  4. package/dist/index.js +1 -1
  5. package/dist/logger.d.ts +11 -0
  6. package/dist/logger.js +26 -0
  7. package/dist/model-cache.d.ts +4 -0
  8. package/dist/model-cache.js +6 -0
  9. package/dist/openrouter-api.d.ts +22 -2
  10. package/dist/openrouter-api.js +21 -2
  11. package/dist/tool-handlers/analyze-audio.d.ts +9 -4
  12. package/dist/tool-handlers/analyze-audio.js +29 -15
  13. package/dist/tool-handlers/analyze-image.d.ts +10 -4
  14. package/dist/tool-handlers/analyze-image.js +37 -9
  15. package/dist/tool-handlers/analyze-video.d.ts +9 -4
  16. package/dist/tool-handlers/analyze-video.js +31 -19
  17. package/dist/tool-handlers/cache.d.ts +33 -0
  18. package/dist/tool-handlers/cache.js +54 -0
  19. package/dist/tool-handlers/chat-completion.d.ts +18 -4
  20. package/dist/tool-handlers/chat-completion.js +39 -11
  21. package/dist/tool-handlers/completion-utils.d.ts +29 -0
  22. package/dist/tool-handlers/completion-utils.js +76 -17
  23. package/dist/tool-handlers/generate-audio.d.ts +2 -0
  24. package/dist/tool-handlers/generate-audio.js +18 -1
  25. package/dist/tool-handlers/generate-image.d.ts +5 -3
  26. package/dist/tool-handlers/generate-image.js +20 -34
  27. package/dist/tool-handlers/generate-video.d.ts +43 -0
  28. package/dist/tool-handlers/generate-video.js +61 -6
  29. package/dist/tool-handlers/get-model-info.d.ts +1 -6
  30. package/dist/tool-handlers/get-model-info.js +2 -1
  31. package/dist/tool-handlers/health-check.d.ts +23 -0
  32. package/dist/tool-handlers/health-check.js +32 -0
  33. package/dist/tool-handlers/openai-withresponse.d.ts +16 -0
  34. package/dist/tool-handlers/openai-withresponse.js +16 -0
  35. package/dist/tool-handlers/path-safety.d.ts +11 -0
  36. package/dist/tool-handlers/path-safety.js +54 -0
  37. package/dist/tool-handlers/provider-routing.js +6 -2
  38. package/dist/tool-handlers/rerank.d.ts +17 -0
  39. package/dist/tool-handlers/rerank.js +52 -0
  40. package/dist/tool-handlers/search-models.d.ts +18 -7
  41. package/dist/tool-handlers/search-models.js +25 -2
  42. package/dist/tool-handlers/structured-output.d.ts +13 -0
  43. package/dist/tool-handlers/structured-output.js +24 -0
  44. package/dist/tool-handlers/validate-model.d.ts +4 -6
  45. package/dist/tool-handlers/validate-model.js +3 -8
  46. package/dist/tool-handlers.d.ts +1 -0
  47. package/dist/tool-handlers.js +417 -165
  48. package/dist/version.d.ts +16 -0
  49. package/dist/version.js +16 -0
  50. package/package.json +1 -1
@@ -11,15 +11,150 @@ import { handleGenerateImage } from './tool-handlers/generate-image.js';
11
11
  import { handleAnalyzeAudio } from './tool-handlers/analyze-audio.js';
12
12
  import { handleGenerateAudio } from './tool-handlers/generate-audio.js';
13
13
  import { handleAnalyzeVideo } from './tool-handlers/analyze-video.js';
14
- import { handleGenerateVideo, handleGetVideoStatus, } from './tool-handlers/generate-video.js';
14
+ import { handleGenerateVideo, handleGetVideoStatus, handleGenerateVideoFromImage, } from './tool-handlers/generate-video.js';
15
+ import { handleRerankDocuments } from './tool-handlers/rerank.js';
16
+ import { handleHealthCheck } from './tool-handlers/health-check.js';
15
17
  function wrapToolArgs(a) {
16
18
  return { params: { arguments: a ?? {} } };
17
19
  }
20
+ function buildProgressHook(server, progressToken) {
21
+ if (progressToken === undefined)
22
+ return undefined;
23
+ return ({ status, progress, attempt, video_id }) => {
24
+ void server.notification({
25
+ method: 'notifications/progress',
26
+ params: {
27
+ progressToken,
28
+ // Progress MUST increase per the MCP spec. We use attempt as a
29
+ // monotonic counter when the upstream doesn't return a numeric
30
+ // progress value.
31
+ progress: typeof progress === 'number' ? progress : attempt,
32
+ ...(typeof progress === 'number' ? { total: 100 } : {}),
33
+ message: `video ${video_id} — ${status}${typeof progress === 'number' ? ` (${progress}%)` : ''}`,
34
+ },
35
+ });
36
+ };
37
+ }
38
+ function extractProgressToken(req) {
39
+ const meta = req?.params
40
+ ?._meta;
41
+ return meta?.progressToken;
42
+ }
43
+ // ---------------------------------------------------------------------------
44
+ // Tool descriptions include explicit "Fails when" and "Works with" sections
45
+ // per arxiv 2602.18764 (Schema-Guided Dialogue / MCP convergence). Explicit
46
+ // failure-mode documentation reduces misrouted calls and helps the model
47
+ // pick the right recovery path after an error.
48
+ const TOOL_DESCRIPTIONS = {
49
+ chat_completion: 'Send messages to an OpenRouter model and get a text response. Supports provider routing ' +
50
+ '(quantizations / ignore / sort / order / require_parameters / data_collection / allow_fallbacks), ' +
51
+ 'model variant suffixes (`:nitro` fastest, `:floor` cheapest, `:exacto` tool-calling accuracy), ' +
52
+ 'reasoning token passthrough, web search, and response caching.\n\n' +
53
+ 'Fails when:\n' +
54
+ '- INVALID_INPUT: messages array is empty\n' +
55
+ '- UPSTREAM_REFUSED: provider rejected the request (credits, content policy, or rate limit)\n' +
56
+ '- UPSTREAM_TIMEOUT: upstream did not respond within the SDK timeout\n' +
57
+ '- MODEL_NOT_FOUND: model slug does not exist on OpenRouter\n\n' +
58
+ 'Works with: validate_model (pre-flight model id check), search_models (discover models).',
59
+ analyze_image: 'Analyze an image with a vision model. Accepts local file paths, http(s) URLs, or base64 data URLs. ' +
60
+ 'Output is model-generated and tagged `_meta.content_is_untrusted: true`.\n\n' +
61
+ 'Fails when:\n' +
62
+ '- INVALID_INPUT: image_path missing or malformed\n' +
63
+ '- UNSAFE_PATH: local path escaped the sandbox\n' +
64
+ '- RESOURCE_TOO_LARGE: image exceeded configured fetch size cap\n' +
65
+ '- UPSTREAM_REFUSED: provider SSRF guard blocked the URL or content policy rejected\n\n' +
66
+ 'Works with: search_models (find vision-capable models), generate_image (follow-up creation).',
67
+ analyze_audio: 'Transcribe or analyze an audio file (WAV / MP3 / FLAC / OGG / etc.) using a multimodal model. ' +
68
+ 'Output is tagged `_meta.content_is_untrusted: true`.\n\n' +
69
+ 'Fails when:\n' +
70
+ '- INVALID_INPUT: audio_path missing\n' +
71
+ '- UNSUPPORTED_FORMAT: decoder could not identify the file as audio\n' +
72
+ '- RESOURCE_TOO_LARGE: input exceeded size cap\n' +
73
+ '- UPSTREAM_REFUSED: blocked host or content policy\n\n' +
74
+ 'Works with: generate_audio (text-to-speech follow-up).',
75
+ analyze_video: 'Describe or analyze a video (mp4 / mpeg / mov / webm) using a multimodal model. Default model: ' +
76
+ 'google/gemini-2.5-flash. Output is tagged `_meta.content_is_untrusted: true`.\n\n' +
77
+ 'Fails when:\n' +
78
+ '- INVALID_INPUT: video_path missing\n' +
79
+ '- UNSUPPORTED_FORMAT: not a recognized video container\n' +
80
+ '- RESOURCE_TOO_LARGE: exceeds fetch cap\n' +
81
+ '- UPSTREAM_REFUSED: SSRF block or provider refusal\n\n' +
82
+ 'Works with: generate_video (text-to-video), get_video_status (poll async jobs).',
83
+ search_models: 'Search OpenRouter\'s model catalog by name, provider, or capability. Returns a paginated list; ' +
84
+ 'use `offset` / `limit` / `next_offset` to page through.\n\n' +
85
+ 'Fails when:\n' +
86
+ '- UPSTREAM_HTTP: /models endpoint returned an error\n' +
87
+ '- UPSTREAM_REFUSED: invalid API key\n\n' +
88
+ 'Works with: validate_model, get_model_info.',
89
+ get_model_info: 'Get pricing / context-length / capability details for a specific model id.\n\n' +
90
+ 'Fails when:\n' +
91
+ '- INVALID_INPUT: model not provided\n' +
92
+ '- MODEL_NOT_FOUND: model slug does not exist\n' +
93
+ '- UPSTREAM_HTTP: model list fetch failed\n\n' +
94
+ 'Works with: search_models (discover ids), validate_model (cheap existence check).',
95
+ validate_model: 'Check whether a model id exists on OpenRouter. Cheap boolean lookup against the cached catalog.\n\n' +
96
+ 'Fails when:\n' +
97
+ '- INVALID_INPUT: model not provided\n' +
98
+ '- UPSTREAM_HTTP: catalog refresh failed\n\n' +
99
+ 'Works with: get_model_info (detailed lookup), chat_completion (pre-flight validation).',
100
+ generate_image: 'Generate an image from a text prompt. Optional reference images condition the output for style / ' +
101
+ 'identity consistency. Default model: google/gemini-2.5-flash-image.\n\n' +
102
+ 'Fails when:\n' +
103
+ '- INVALID_INPUT: prompt empty, bad aspect_ratio / image_size, unreadable reference image\n' +
104
+ '- UNSAFE_PATH: save_path or input_images path escaped the sandbox\n' +
105
+ '- UPSTREAM_REFUSED: provider content policy rejected or insufficient credits\n' +
106
+ '- MODEL_NOT_FOUND: model slug invalid\n\n' +
107
+ 'Works with: analyze_image (verify the result), generate_video_from_image (next step in workflow).',
108
+ generate_audio: 'Generate audio (speech / music) from a text prompt. Format auto-detected, extension auto-corrected.\n\n' +
109
+ 'Fails when:\n' +
110
+ '- INVALID_INPUT: prompt empty\n' +
111
+ '- UNSAFE_PATH: save_path escaped the sandbox\n' +
112
+ '- UPSTREAM_REFUSED: content policy or credit issues\n\n' +
113
+ 'Works with: analyze_audio (verify the result).',
114
+ generate_video: 'Generate a video from a text prompt (optionally conditioned on first/last-frame or reference images). ' +
115
+ 'Submits an async job, polls until completion or max_wait_ms, and downloads the result. Emits MCP ' +
116
+ 'progress notifications when the client provides a `progressToken`. Default model: google/veo-3.1.\n\n' +
117
+ 'Fails when:\n' +
118
+ '- INVALID_INPUT: prompt empty\n' +
119
+ '- UNSAFE_PATH: save_path or reference image paths escaped the sandbox\n' +
120
+ '- UPSTREAM_REFUSED: content policy, credits, or bad request\n' +
121
+ '- JOB_FAILED: provider marked the job as failed\n' +
122
+ '- JOB_STILL_RUNNING: exceeded max_wait_ms (response carries the video_id to resume)\n' +
123
+ '- UNSUPPORTED_FORMAT: reference/frame image could not be decoded\n\n' +
124
+ 'Works with: get_video_status (resume timed-out jobs), generate_video_from_image (narrower image-to-video variant).',
125
+ generate_video_from_image: 'Narrower convenience wrapper around generate_video for image-to-video workflows. Takes a single ' +
126
+ '`image` argument (used as the first frame) and `prompt`. Per arxiv 2511.03497, narrower tools with ' +
127
+ 'fewer parameters improve tool-call hit rate.\n\n' +
128
+ 'Fails when:\n' +
129
+ '- INVALID_INPUT: image or prompt missing\n' +
130
+ '- UNSAFE_PATH: image path escaped the sandbox\n' +
131
+ '- UPSTREAM_REFUSED / JOB_FAILED / JOB_STILL_RUNNING: same as generate_video\n\n' +
132
+ 'Works with: generate_video (full parameter surface), get_video_status.',
133
+ get_video_status: 'Poll an async video-generation job by id. Downloads the result when complete (and saves if save_path given).\n\n' +
134
+ 'Fails when:\n' +
135
+ '- INVALID_INPUT: video_id missing\n' +
136
+ '- UNSAFE_PATH: save_path escaped the sandbox\n' +
137
+ '- JOB_FAILED: provider marked the job as failed\n' +
138
+ '- JOB_STILL_RUNNING: job not yet complete (carries `_meta.last_status` + `progress`)\n\n' +
139
+ 'Works with: generate_video, generate_video_from_image.',
140
+ rerank_documents: 'Re-order a list of documents by relevance to a query using an OpenRouter reranker. Default model: ' +
141
+ 'cohere/rerank-english-v3.0.\n\n' +
142
+ 'Fails when:\n' +
143
+ '- INVALID_INPUT: query missing, documents empty, non-string document elements\n' +
144
+ '- MODEL_NOT_FOUND: reranker model slug does not exist\n' +
145
+ '- UPSTREAM_HTTP: provider returned an error\n\n' +
146
+ 'Works with: search_models (discover rerankers), chat_completion (answer grounded in top-ranked docs).',
147
+ health_check: 'Verify API-key validity, OpenRouter reachability, and return server + protocol versions. No args.\n\n' +
148
+ 'Fails when: never returns an error result — always returns `{ ok, api_key_valid, ... }` so ops can ' +
149
+ 'programmatically branch on the payload.\n\n' +
150
+ 'Works with: every other tool (run once at startup to confirm credentials).',
151
+ };
18
152
  export class ToolHandlers {
19
153
  openai;
20
154
  modelCache = ModelCache.getInstance();
21
155
  apiClient;
22
156
  defaultModel;
157
+ server;
23
158
  constructor(server, apiKey, defaultModel) {
24
159
  this.defaultModel = defaultModel;
25
160
  this.apiClient = new OpenRouterAPIClient(apiKey);
@@ -27,6 +162,7 @@ export class ToolHandlers {
27
162
  apiKey,
28
163
  baseURL: 'https://openrouter.ai/api/v1',
29
164
  });
165
+ this.server = server;
30
166
  this.register(server);
31
167
  }
32
168
  register(server) {
@@ -34,18 +170,22 @@ export class ToolHandlers {
34
170
  tools: [
35
171
  {
36
172
  name: 'chat_completion',
37
- description: 'Send messages to an OpenRouter model and get a response. Supports provider routing (quantizations / ignore / sort / order / require_parameters / data_collection / allow_fallbacks) and model variant suffixes (`:nitro` for faster, `:floor` for cheapest).',
173
+ description: TOOL_DESCRIPTIONS.chat_completion,
38
174
  annotations: {
175
+ title: 'Chat completion',
39
176
  readOnlyHint: false,
40
177
  destructiveHint: false,
41
178
  idempotentHint: false,
179
+ openWorldHint: true,
42
180
  },
43
181
  inputSchema: {
44
182
  type: 'object',
45
183
  properties: {
46
184
  model: {
47
185
  type: 'string',
48
- description: 'Model ID (optional, uses default). Append `:nitro` for faster/experimental variants or `:floor` for the cheapest available variant (e.g. `openai/gpt-4o:nitro`).',
186
+ description: 'Model ID (optional, uses default). Append `:nitro` for the fastest variant, ' +
187
+ '`:floor` for the cheapest, or `:exacto` for the best tool-calling accuracy. ' +
188
+ 'Example: `openai/gpt-4o:nitro`.',
49
189
  },
50
190
  messages: {
51
191
  type: 'array',
@@ -69,7 +209,8 @@ export class ToolHandlers {
69
209
  },
70
210
  provider: {
71
211
  type: 'object',
72
- description: 'OpenRouter provider-routing overrides. Merges on top of `OPENROUTER_PROVIDER_*` env defaults. See https://openrouter.ai/docs/features/provider-routing',
212
+ description: 'OpenRouter provider-routing overrides. Merges on top of `OPENROUTER_PROVIDER_*` env defaults. ' +
213
+ 'See https://openrouter.ai/docs/features/provider-routing',
73
214
  properties: {
74
215
  quantizations: {
75
216
  type: 'array',
@@ -79,44 +220,57 @@ export class ToolHandlers {
79
220
  ignore: {
80
221
  type: 'array',
81
222
  items: { type: 'string' },
82
- description: 'Exclude these provider slugs (e.g. `["openai","anthropic"]`).',
223
+ description: 'Exclude these provider slugs.',
83
224
  },
84
225
  sort: {
85
226
  type: 'string',
86
227
  enum: ['price', 'throughput', 'latency'],
87
- description: 'Sort providers by this criterion.',
88
- },
89
- order: {
90
- type: 'array',
91
- items: { type: 'string' },
92
- description: 'Prioritized list of provider IDs (e.g. `["openai/gpt-4o","anthropic/claude-3-opus"]`).',
93
- },
94
- require_parameters: {
95
- type: 'boolean',
96
- description: 'Only use providers that support every parameter in the request.',
97
- },
98
- data_collection: {
99
- type: 'string',
100
- enum: ['allow', 'deny'],
101
- description: 'Whether providers may collect request data.',
102
- },
103
- allow_fallbacks: {
104
- type: 'boolean',
105
- description: 'Allow fallback to unlisted providers when preferred ones fail.',
106
228
  },
229
+ order: { type: 'array', items: { type: 'string' } },
230
+ require_parameters: { type: 'boolean' },
231
+ data_collection: { type: 'string', enum: ['allow', 'deny'] },
232
+ allow_fallbacks: { type: 'boolean' },
107
233
  },
108
234
  },
235
+ include_reasoning: {
236
+ type: 'boolean',
237
+ description: 'Surface the model\'s chain-of-thought on `_meta.reasoning` for R1 / Opus 4.7 / Gemini Thinking.',
238
+ },
239
+ online: {
240
+ type: 'boolean',
241
+ description: 'Enable OpenRouter\'s web-search plugin (Exa-backed, $4 / 1000 results).',
242
+ },
243
+ web_max_results: {
244
+ type: 'number',
245
+ minimum: 1,
246
+ description: 'Max web-search results when `online: true` (default 5).',
247
+ },
248
+ cache: {
249
+ type: 'boolean',
250
+ description: 'Enable OpenRouter response caching via `X-OpenRouter-Cache: true`. ' +
251
+ 'Server-wide default settable via `OPENROUTER_CACHE_RESPONSES=1`.',
252
+ },
253
+ cache_ttl: {
254
+ type: 'string',
255
+ description: 'Cache TTL (e.g. `"5m"`, `"1h"`, `"24h"`; 1s-24h range).',
256
+ },
257
+ cache_clear: {
258
+ type: 'boolean',
259
+ description: 'Bust the cache entry for this exact request.',
260
+ },
109
261
  },
110
262
  required: ['messages'],
111
263
  },
112
264
  },
113
265
  {
114
266
  name: 'analyze_image',
115
- description: 'Analyze an image using a vision model',
267
+ description: TOOL_DESCRIPTIONS.analyze_image,
116
268
  annotations: {
269
+ title: 'Analyze image',
117
270
  readOnlyHint: true,
118
271
  destructiveHint: false,
119
272
  idempotentHint: false,
273
+ openWorldHint: true,
120
274
  },
121
275
  inputSchema: {
122
276
  type: 'object',
@@ -124,17 +278,27 @@ export class ToolHandlers {
124
278
  image_path: { type: 'string', description: 'File path, URL, or data URL' },
125
279
  question: { type: 'string', description: 'Question about the image' },
126
280
  model: { type: 'string' },
281
+ cache_input: {
282
+ type: 'boolean',
283
+ description: 'Attach `cache_control: ephemeral` to the image block so Anthropic / Gemini prompt-cache it. ' +
284
+ 'Repeat questions about the same image save ~10x on Anthropic.',
285
+ },
286
+ cache: { type: 'boolean' },
287
+ cache_ttl: { type: 'string' },
288
+ cache_clear: { type: 'boolean' },
127
289
  },
128
290
  required: ['image_path'],
129
291
  },
130
292
  },
131
293
  {
132
294
  name: 'analyze_audio',
133
- description: 'Analyze or transcribe an audio file using a multimodal model',
295
+ description: TOOL_DESCRIPTIONS.analyze_audio,
134
296
  annotations: {
297
+ title: 'Analyze audio',
135
298
  readOnlyHint: true,
136
299
  destructiveHint: false,
137
300
  idempotentHint: false,
301
+ openWorldHint: true,
138
302
  },
139
303
  inputSchema: {
140
304
  type: 'object',
@@ -148,41 +312,50 @@ export class ToolHandlers {
148
312
  description: 'Question or instruction about the audio (default: transcribe)',
149
313
  },
150
314
  model: { type: 'string' },
315
+ cache_input: { type: 'boolean' },
316
+ cache: { type: 'boolean' },
317
+ cache_ttl: { type: 'string' },
318
+ cache_clear: { type: 'boolean' },
151
319
  },
152
320
  required: ['audio_path'],
153
321
  },
154
322
  },
155
323
  {
156
324
  name: 'analyze_video',
157
- description: 'Analyze or transcribe a video file using a multimodal model. Accepts mp4, mpeg, mov, or webm from a local file path, HTTP(S) URL, or base64 data URL. Default model: google/gemini-2.5-flash.',
325
+ description: TOOL_DESCRIPTIONS.analyze_video,
158
326
  annotations: {
327
+ title: 'Analyze video',
159
328
  readOnlyHint: true,
160
329
  destructiveHint: false,
161
330
  idempotentHint: false,
331
+ openWorldHint: true,
162
332
  },
163
333
  inputSchema: {
164
334
  type: 'object',
165
335
  properties: {
166
336
  video_path: {
167
337
  type: 'string',
168
- description: 'File path, HTTP(S) URL, or base64 data URL. Supported formats: mp4, mpeg, mov, webm.',
338
+ description: 'File path, HTTP(S) URL, or base64 data URL. Supported: mp4 / mpeg / mov / webm.',
169
339
  },
170
- question: {
171
- type: 'string',
172
- description: 'Question or instruction about the video (default: describe).',
173
- },
174
- model: { type: 'string', description: 'Override the model ID.' },
340
+ question: { type: 'string' },
341
+ model: { type: 'string' },
342
+ cache_input: { type: 'boolean' },
343
+ cache: { type: 'boolean' },
344
+ cache_ttl: { type: 'string' },
345
+ cache_clear: { type: 'boolean' },
175
346
  },
176
347
  required: ['video_path'],
177
348
  },
178
349
  },
179
350
  {
180
351
  name: 'search_models',
181
- description: 'Search available OpenRouter models',
352
+ description: TOOL_DESCRIPTIONS.search_models,
182
353
  annotations: {
354
+ title: 'Search models',
183
355
  readOnlyHint: true,
184
356
  destructiveHint: false,
185
357
  idempotentHint: true,
358
+ openWorldHint: true,
186
359
  },
187
360
  inputSchema: {
188
361
  type: 'object',
@@ -198,47 +371,81 @@ export class ToolHandlers {
198
371
  },
199
372
  },
200
373
  limit: { type: 'number', minimum: 1, maximum: 50 },
374
+ offset: { type: 'number', minimum: 0 },
201
375
  },
202
376
  },
377
+ outputSchema: {
378
+ type: 'object',
379
+ properties: {
380
+ results: { type: 'array', items: { type: 'object' } },
381
+ offset: { type: 'number' },
382
+ limit: { type: 'number' },
383
+ total: { type: 'number' },
384
+ has_more: { type: 'boolean' },
385
+ next_offset: { type: ['number', 'null'] },
386
+ },
387
+ required: ['results', 'offset', 'limit', 'total', 'has_more', 'next_offset'],
388
+ },
203
389
  },
204
390
  {
205
391
  name: 'get_model_info',
206
- description: 'Get details about a specific model',
392
+ description: TOOL_DESCRIPTIONS.get_model_info,
207
393
  annotations: {
394
+ title: 'Get model info',
208
395
  readOnlyHint: true,
209
396
  destructiveHint: false,
210
397
  idempotentHint: true,
398
+ openWorldHint: true,
211
399
  },
212
400
  inputSchema: {
213
401
  type: 'object',
214
402
  properties: { model: { type: 'string' } },
215
403
  required: ['model'],
216
404
  },
405
+ outputSchema: {
406
+ type: 'object',
407
+ properties: {
408
+ id: { type: 'string' },
409
+ name: { type: 'string' },
410
+ context_length: { type: 'number' },
411
+ architecture: { type: 'object' },
412
+ },
413
+ required: ['id'],
414
+ },
217
415
  },
218
416
  {
219
417
  name: 'validate_model',
220
- description: 'Check if a model ID exists',
418
+ description: TOOL_DESCRIPTIONS.validate_model,
221
419
  annotations: {
420
+ title: 'Validate model',
222
421
  readOnlyHint: true,
223
422
  destructiveHint: false,
224
423
  idempotentHint: true,
424
+ openWorldHint: true,
225
425
  },
226
426
  inputSchema: {
227
427
  type: 'object',
228
428
  properties: { model: { type: 'string' } },
229
429
  required: ['model'],
230
430
  },
431
+ outputSchema: {
432
+ type: 'object',
433
+ properties: {
434
+ valid: { type: 'boolean' },
435
+ model: { type: 'string' },
436
+ },
437
+ required: ['valid', 'model'],
438
+ },
231
439
  },
232
440
  {
233
441
  name: 'generate_image',
234
- description: 'Generate an image from a text prompt. Optionally conditioned on one or more ' +
235
- 'reference images (file paths, http(s) URLs, or data URLs) for character / style ' +
236
- 'consistency. Sends `modalities: ["image","text"]` by default; override via the ' +
237
- '`modalities` field if needed.',
442
+ description: TOOL_DESCRIPTIONS.generate_image,
238
443
  annotations: {
444
+ title: 'Generate image',
239
445
  readOnlyHint: false,
240
446
  destructiveHint: false,
241
447
  idempotentHint: false,
448
+ openWorldHint: true,
242
449
  },
243
450
  inputSchema: {
244
451
  type: 'object',
@@ -247,7 +454,6 @@ export class ToolHandlers {
247
454
  model: { type: 'string' },
248
455
  aspect_ratio: {
249
456
  type: 'string',
250
- description: 'Output aspect ratio (e.g. 1:1, 16:9, 9:16, 4:3, 3:4, 21:9). Model-dependent.',
251
457
  enum: [
252
458
  '1:1',
253
459
  '2:3',
@@ -265,181 +471,227 @@ export class ToolHandlers {
265
471
  '8:1',
266
472
  ],
267
473
  },
268
- image_size: {
269
- type: 'string',
270
- description: 'Output resolution bucket. 1K is the default; 0.5K / 2K / 4K are model-dependent.',
271
- enum: ['0.5K', '1K', '2K', '4K'],
272
- },
273
- max_tokens: {
274
- type: 'number',
275
- minimum: 1,
276
- description: 'Cap on completion tokens. Defaults to the model context window, which can trip free-tier quotas; set e.g. 4096 on low-credit accounts.',
277
- },
278
- save_path: {
279
- type: 'string',
280
- description: 'Optional path to save the image. Routed through the OPENROUTER_OUTPUT_DIR sandbox.',
281
- },
282
- input_images: {
283
- type: 'array',
284
- items: { type: 'string' },
285
- description: 'Optional reference images for visual consistency. Each entry may be a ' +
286
- 'local file path (sandboxed to OPENROUTER_INPUT_DIR / OPENROUTER_OUTPUT_DIR / ' +
287
- 'cwd), an http(s) URL, or a `data:image/...;base64,...` URL. Inlined as ' +
288
- 'multimodal user content in the order given.',
289
- },
290
- modalities: {
291
- type: 'array',
292
- items: { type: 'string' },
293
- description: 'Override the default `modalities: ["image","text"]` sent to OpenRouter. ' +
294
- 'Most callers should leave this unset. Provide e.g. ["text"] to suppress ' +
295
- 'image output for inspection / captioning.',
296
- },
474
+ image_size: { type: 'string', enum: ['0.5K', '1K', '2K', '4K'] },
475
+ max_tokens: { type: 'number', minimum: 1 },
476
+ save_path: { type: 'string' },
477
+ input_images: { type: 'array', items: { type: 'string' } },
478
+ modalities: { type: 'array', items: { type: 'string' } },
297
479
  },
298
480
  required: ['prompt'],
299
481
  },
300
482
  },
301
483
  {
302
484
  name: 'generate_audio',
303
- description: 'Generate audio from a text prompt. Conversational models (e.g. openai/gpt-audio) respond in spoken audio. Music models (e.g. google/lyria-3-clip-preview) need a structured prompt. Output format is auto-detected and file extension is corrected automatically.',
485
+ description: TOOL_DESCRIPTIONS.generate_audio,
304
486
  annotations: {
487
+ title: 'Generate audio',
305
488
  readOnlyHint: false,
306
489
  destructiveHint: false,
307
490
  idempotentHint: false,
491
+ openWorldHint: true,
308
492
  },
309
493
  inputSchema: {
310
494
  type: 'object',
311
495
  properties: {
312
- prompt: { type: 'string', description: 'Text input' },
313
- model: { type: 'string', description: 'Model ID (default: openai/gpt-audio)' },
314
- voice: { type: 'string', description: 'Voice name (default: alloy)' },
315
- format: {
316
- type: 'string',
317
- description: 'Requested format: pcm16 (default), mp3, flac, opus',
318
- },
319
- save_path: {
320
- type: 'string',
321
- description: 'Optional path to save the audio. Extension auto-corrected and routed through OPENROUTER_OUTPUT_DIR sandbox.',
322
- },
496
+ prompt: { type: 'string' },
497
+ model: { type: 'string' },
498
+ voice: { type: 'string' },
499
+ format: { type: 'string' },
500
+ save_path: { type: 'string' },
323
501
  },
324
502
  required: ['prompt'],
325
503
  },
326
504
  },
327
505
  {
328
506
  name: 'generate_video',
329
- description: 'Generate a video from a text prompt using an OpenRouter video-generation model (default: google/veo-3.1). ' +
330
- 'Submits an async job, polls until completion or max_wait_ms, then downloads the result. ' +
331
- 'Optionally conditioned on first/last-frame images or reference images. ' +
332
- 'Large outputs are auto-saved when save_path is provided and path-sandboxed.',
507
+ description: TOOL_DESCRIPTIONS.generate_video,
333
508
  annotations: {
509
+ title: 'Generate video',
334
510
  readOnlyHint: false,
335
511
  destructiveHint: false,
336
512
  idempotentHint: false,
513
+ openWorldHint: true,
337
514
  },
338
515
  inputSchema: {
339
516
  type: 'object',
340
517
  properties: {
341
- prompt: { type: 'string', description: 'Text description of the desired video.' },
342
- model: { type: 'string', description: 'Override the video model ID.' },
343
- resolution: {
344
- type: 'string',
345
- description: '480p / 720p / 1080p / 1K / 2K / 4K (model-dependent).',
346
- },
347
- aspect_ratio: {
348
- type: 'string',
349
- description: '16:9 / 9:16 / 1:1 / 4:3 / 3:4 / 21:9 / 9:21 (model-dependent).',
350
- },
351
- duration: {
352
- type: 'number',
353
- minimum: 1,
354
- description: 'Duration in seconds (model-dependent).',
355
- },
356
- seed: { type: 'number', description: 'Deterministic seed when supported.' },
357
- first_frame_image: {
358
- type: 'string',
359
- description: 'Optional image (path, URL, or data URL) used as the first frame for image-to-video.',
360
- },
361
- last_frame_image: {
362
- type: 'string',
363
- description: 'Optional image used as the last frame for frame transitions.',
364
- },
365
- reference_images: {
366
- type: 'array',
367
- items: { type: 'string' },
368
- description: 'Optional style/content reference images.',
369
- },
370
- provider: {
371
- type: 'object',
372
- description: 'Provider-specific passthrough options keyed by provider slug.',
373
- },
374
- save_path: {
518
+ prompt: { type: 'string' },
519
+ model: { type: 'string' },
520
+ resolution: { type: 'string' },
521
+ aspect_ratio: { type: 'string' },
522
+ duration: { type: 'number', minimum: 1 },
523
+ seed: { type: 'number' },
524
+ first_frame_image: { type: 'string' },
525
+ last_frame_image: { type: 'string' },
526
+ reference_images: { type: 'array', items: { type: 'string' } },
527
+ provider: { type: 'object' },
528
+ save_path: { type: 'string' },
529
+ max_wait_ms: { type: 'number', minimum: 10000 },
530
+ poll_interval_ms: { type: 'number', minimum: 2000 },
531
+ },
532
+ required: ['prompt'],
533
+ },
534
+ },
535
+ {
536
+ name: 'generate_video_from_image',
537
+ description: TOOL_DESCRIPTIONS.generate_video_from_image,
538
+ annotations: {
539
+ title: 'Generate video from image',
540
+ readOnlyHint: false,
541
+ destructiveHint: false,
542
+ idempotentHint: false,
543
+ openWorldHint: true,
544
+ },
545
+ inputSchema: {
546
+ type: 'object',
547
+ properties: {
548
+ image: {
375
549
  type: 'string',
376
- description: 'Where to save the video. Routed through the OPENROUTER_OUTPUT_DIR sandbox; extension auto-corrected.',
377
- },
378
- max_wait_ms: {
379
- type: 'number',
380
- minimum: 10000,
381
- description: 'Total time to wait for the async job before returning a resumable handle (default 600000 ms).',
382
- },
383
- poll_interval_ms: {
384
- type: 'number',
385
- minimum: 2000,
386
- description: 'Polling cadence (default 15000 ms).',
550
+ description: 'First-frame image (path, URL, or data URL). Required.',
387
551
  },
552
+ prompt: { type: 'string' },
553
+ model: { type: 'string' },
554
+ resolution: { type: 'string' },
555
+ aspect_ratio: { type: 'string' },
556
+ duration: { type: 'number', minimum: 1 },
557
+ seed: { type: 'number' },
558
+ save_path: { type: 'string' },
559
+ max_wait_ms: { type: 'number', minimum: 10000 },
560
+ poll_interval_ms: { type: 'number', minimum: 2000 },
388
561
  },
389
- required: ['prompt'],
562
+ required: ['image', 'prompt'],
390
563
  },
391
564
  },
392
565
  {
393
566
  name: 'get_video_status',
394
- description: 'Resume a previously submitted video generation job by id. Returns the latest status; if completed, ' +
395
- 'downloads the video (and saves it when save_path is provided).',
567
+ description: TOOL_DESCRIPTIONS.get_video_status,
396
568
  annotations: {
569
+ title: 'Get video status',
397
570
  readOnlyHint: true,
398
571
  destructiveHint: false,
399
572
  idempotentHint: true,
573
+ openWorldHint: true,
400
574
  },
401
575
  inputSchema: {
402
576
  type: 'object',
403
577
  properties: {
404
- video_id: { type: 'string', description: 'Job id from a previous generate_video call.' },
405
- save_path: {
406
- type: 'string',
407
- description: 'Optional save path (applies when the job is already completed).',
408
- },
578
+ video_id: { type: 'string' },
579
+ save_path: { type: 'string' },
409
580
  },
410
581
  required: ['video_id'],
411
582
  },
412
583
  },
584
+ {
585
+ name: 'rerank_documents',
586
+ description: TOOL_DESCRIPTIONS.rerank_documents,
587
+ annotations: {
588
+ title: 'Rerank documents',
589
+ readOnlyHint: true,
590
+ destructiveHint: false,
591
+ idempotentHint: true,
592
+ openWorldHint: true,
593
+ },
594
+ inputSchema: {
595
+ type: 'object',
596
+ properties: {
597
+ query: { type: 'string' },
598
+ documents: { type: 'array', items: { type: 'string' }, minItems: 1 },
599
+ model: { type: 'string' },
600
+ top_n: { type: 'number', minimum: 1 },
601
+ return_documents: { type: 'boolean' },
602
+ },
603
+ required: ['query', 'documents'],
604
+ },
605
+ outputSchema: {
606
+ type: 'object',
607
+ properties: {
608
+ model: { type: 'string' },
609
+ results: {
610
+ type: 'array',
611
+ items: {
612
+ type: 'object',
613
+ properties: {
614
+ index: { type: 'number' },
615
+ score: { type: 'number' },
616
+ document: { type: 'string' },
617
+ },
618
+ required: ['index', 'score'],
619
+ },
620
+ },
621
+ },
622
+ required: ['results'],
623
+ },
624
+ },
625
+ {
626
+ name: 'health_check',
627
+ description: TOOL_DESCRIPTIONS.health_check,
628
+ annotations: {
629
+ title: 'Health check',
630
+ readOnlyHint: true,
631
+ destructiveHint: false,
632
+ idempotentHint: true,
633
+ openWorldHint: true,
634
+ },
635
+ inputSchema: { type: 'object', properties: {} },
636
+ outputSchema: {
637
+ type: 'object',
638
+ properties: {
639
+ ok: { type: 'boolean' },
640
+ server_version: { type: 'string' },
641
+ protocol_version: { type: 'string' },
642
+ api_key_valid: { type: 'boolean' },
643
+ models_cached: { type: 'number' },
644
+ error: { type: 'string' },
645
+ },
646
+ required: ['ok', 'server_version', 'protocol_version', 'api_key_valid', 'models_cached'],
647
+ },
648
+ },
413
649
  ],
414
650
  }));
415
651
  server.setRequestHandler(CallToolRequestSchema, async (request) => {
416
652
  const { name, arguments: args } = request.params;
417
- switch (name) {
418
- case 'chat_completion':
419
- return handleChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
420
- case 'analyze_image':
421
- return handleAnalyzeImage(wrapToolArgs(args), this.openai, this.defaultModel);
422
- case 'analyze_audio':
423
- return handleAnalyzeAudio(wrapToolArgs(args), this.openai, this.defaultModel);
424
- case 'analyze_video':
425
- return handleAnalyzeVideo(wrapToolArgs(args), this.openai, this.defaultModel);
426
- case 'search_models':
427
- return handleSearchModels(wrapToolArgs(args), this.apiClient, this.modelCache);
428
- case 'get_model_info':
429
- return handleGetModelInfo(wrapToolArgs(args), this.modelCache, this.apiClient);
430
- case 'validate_model':
431
- return handleValidateModel(wrapToolArgs(args), this.modelCache, this.apiClient);
432
- case 'generate_image':
433
- return handleGenerateImage(wrapToolArgs(args), this.openai);
434
- case 'generate_audio':
435
- return handleGenerateAudio(wrapToolArgs(args), this.openai);
436
- case 'generate_video':
437
- return handleGenerateVideo(wrapToolArgs(args), this.apiClient);
438
- case 'get_video_status':
439
- return handleGetVideoStatus(wrapToolArgs(args), this.apiClient);
440
- default:
441
- throw new McpError(McpErrorCode.MethodNotFound, `Unknown tool: ${name}`);
442
- }
653
+ // Our handlers return structured shapes that satisfy CallToolResult's
654
+ // open interface (content + optional _meta + optional isError +
655
+ // optional structuredContent). We cast through unknown because
656
+ // several handlers include server-specific _meta keys (e.g.
657
+ // server_version, cache, code) that aren't listed in the SDK's
658
+ // typed schema — the SDK accepts any extras thanks to the
659
+ // `[x: string]: unknown` index signature.
660
+ const dispatch = async () => {
661
+ switch (name) {
662
+ case 'chat_completion':
663
+ return handleChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
664
+ case 'analyze_image':
665
+ return handleAnalyzeImage(wrapToolArgs(args), this.openai, this.defaultModel);
666
+ case 'analyze_audio':
667
+ return handleAnalyzeAudio(wrapToolArgs(args), this.openai, this.defaultModel);
668
+ case 'analyze_video':
669
+ return handleAnalyzeVideo(wrapToolArgs(args), this.openai, this.defaultModel);
670
+ case 'search_models':
671
+ return handleSearchModels(wrapToolArgs(args), this.apiClient, this.modelCache);
672
+ case 'get_model_info':
673
+ return handleGetModelInfo(wrapToolArgs(args), this.modelCache, this.apiClient);
674
+ case 'validate_model':
675
+ return handleValidateModel(wrapToolArgs(args), this.modelCache, this.apiClient);
676
+ case 'generate_image':
677
+ return handleGenerateImage(wrapToolArgs(args), this.openai);
678
+ case 'generate_audio':
679
+ return handleGenerateAudio(wrapToolArgs(args), this.openai);
680
+ case 'generate_video':
681
+ return handleGenerateVideo(wrapToolArgs(args), this.apiClient, buildProgressHook(this.server, extractProgressToken(request)));
682
+ case 'generate_video_from_image':
683
+ return handleGenerateVideoFromImage(wrapToolArgs(args), this.apiClient, buildProgressHook(this.server, extractProgressToken(request)));
684
+ case 'get_video_status':
685
+ return handleGetVideoStatus(wrapToolArgs(args), this.apiClient);
686
+ case 'rerank_documents':
687
+ return handleRerankDocuments(wrapToolArgs(args), this.apiClient);
688
+ case 'health_check':
689
+ return handleHealthCheck(wrapToolArgs(args), this.apiClient, this.modelCache);
690
+ default:
691
+ throw new McpError(McpErrorCode.MethodNotFound, `Unknown tool: ${name}`);
692
+ }
693
+ };
694
+ return (await dispatch());
443
695
  });
444
696
  }
445
697
  }