@stabgan/openrouter-mcp-multimodal 4.0.1 → 4.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +95 -8
  2. package/dist/errors.d.ts +26 -8
  3. package/dist/errors.js +17 -12
  4. package/dist/index.js +21 -5
  5. package/dist/logger.d.ts +11 -0
  6. package/dist/logger.js +26 -0
  7. package/dist/model-cache.d.ts +17 -0
  8. package/dist/model-cache.js +27 -1
  9. package/dist/openrouter-api.d.ts +22 -2
  10. package/dist/openrouter-api.js +21 -2
  11. package/dist/tool-handlers/analyze-audio.d.ts +9 -4
  12. package/dist/tool-handlers/analyze-audio.js +29 -15
  13. package/dist/tool-handlers/analyze-image.d.ts +10 -4
  14. package/dist/tool-handlers/analyze-image.js +37 -9
  15. package/dist/tool-handlers/analyze-video.d.ts +9 -4
  16. package/dist/tool-handlers/analyze-video.js +31 -19
  17. package/dist/tool-handlers/cache.d.ts +33 -0
  18. package/dist/tool-handlers/cache.js +54 -0
  19. package/dist/tool-handlers/chat-completion.d.ts +18 -4
  20. package/dist/tool-handlers/chat-completion.js +39 -11
  21. package/dist/tool-handlers/completion-utils.d.ts +29 -0
  22. package/dist/tool-handlers/completion-utils.js +76 -17
  23. package/dist/tool-handlers/generate-audio.d.ts +2 -0
  24. package/dist/tool-handlers/generate-audio.js +18 -1
  25. package/dist/tool-handlers/generate-image.d.ts +2 -0
  26. package/dist/tool-handlers/generate-image.js +15 -0
  27. package/dist/tool-handlers/generate-video.d.ts +43 -0
  28. package/dist/tool-handlers/generate-video.js +46 -3
  29. package/dist/tool-handlers/get-model-info.d.ts +1 -6
  30. package/dist/tool-handlers/get-model-info.js +2 -1
  31. package/dist/tool-handlers/health-check.d.ts +23 -0
  32. package/dist/tool-handlers/health-check.js +35 -0
  33. package/dist/tool-handlers/openai-withresponse.d.ts +16 -0
  34. package/dist/tool-handlers/openai-withresponse.js +16 -0
  35. package/dist/tool-handlers/openrouter-errors.d.ts +5 -1
  36. package/dist/tool-handlers/openrouter-errors.js +72 -11
  37. package/dist/tool-handlers/rerank.d.ts +17 -0
  38. package/dist/tool-handlers/rerank.js +52 -0
  39. package/dist/tool-handlers/search-models.d.ts +18 -7
  40. package/dist/tool-handlers/search-models.js +25 -2
  41. package/dist/tool-handlers/structured-output.d.ts +13 -0
  42. package/dist/tool-handlers/structured-output.js +24 -0
  43. package/dist/tool-handlers/validate-model.d.ts +4 -6
  44. package/dist/tool-handlers/validate-model.js +3 -8
  45. package/dist/tool-handlers.d.ts +1 -0
  46. package/dist/tool-handlers.js +435 -165
  47. package/dist/version.d.ts +16 -0
  48. package/dist/version.js +16 -0
  49. package/package.json +1 -1
@@ -11,15 +11,168 @@ import { handleGenerateImage } from './tool-handlers/generate-image.js';
11
11
  import { handleAnalyzeAudio } from './tool-handlers/analyze-audio.js';
12
12
  import { handleGenerateAudio } from './tool-handlers/generate-audio.js';
13
13
  import { handleAnalyzeVideo } from './tool-handlers/analyze-video.js';
14
- import { handleGenerateVideo, handleGetVideoStatus, } from './tool-handlers/generate-video.js';
14
+ import { handleGenerateVideo, handleGetVideoStatus, handleGenerateVideoFromImage, } from './tool-handlers/generate-video.js';
15
+ import { handleRerankDocuments } from './tool-handlers/rerank.js';
16
+ import { handleHealthCheck } from './tool-handlers/health-check.js';
15
17
  function wrapToolArgs(a) {
16
18
  return { params: { arguments: a ?? {} } };
17
19
  }
20
+ function buildProgressHook(server, progressToken) {
21
+ if (progressToken === undefined)
22
+ return undefined;
23
+ // MCP `notifications/progress` REQUIRES `progress` to be strictly
24
+ // monotonically increasing within a single progressToken. OpenRouter
25
+ // returns `progress: 0..100` on some ticks and omits it on others, so
26
+ // we anchor on a per-hook attempt counter and use the upstream number
27
+ // only as an informational `message`. This guarantees monotonicity
28
+ // regardless of what the upstream does (drops, duplicates, decreases).
29
+ //
30
+ // See MCP spec 2025-06-18 utilities/progress §Behavior Requirements:
31
+ // "The progress value MUST increase with each notification, even if
32
+ // the total is unknown."
33
+ let lastSent = -1;
34
+ return ({ status, progress, attempt, video_id }) => {
35
+ // Always monotonic: at least attempt+1 (so initial attempt=0 → 0 stays
36
+ // reserved for the 'submitted' ping). If upstream has a real numeric
37
+ // progress that's higher than our counter, adopt that.
38
+ const candidate = typeof progress === 'number' ? Math.max(attempt, progress) : attempt;
39
+ const next = Math.max(lastSent + 1, candidate);
40
+ lastSent = next;
41
+ void server.notification({
42
+ method: 'notifications/progress',
43
+ params: {
44
+ progressToken,
45
+ progress: next,
46
+ message: `video ${video_id} — ${status}${typeof progress === 'number' ? ` (${progress}%)` : ''}`,
47
+ },
48
+ });
49
+ };
50
+ }
51
+ function extractProgressToken(req) {
52
+ const meta = req?.params
53
+ ?._meta;
54
+ return meta?.progressToken;
55
+ }
56
+ // ---------------------------------------------------------------------------
57
+ // Tool descriptions include explicit "Fails when" and "Works with" sections
58
+ // per arxiv 2602.18764 (Schema-Guided Dialogue / MCP convergence). Explicit
59
+ // failure-mode documentation reduces misrouted calls and helps the model
60
+ // pick the right recovery path after an error.
61
+ const TOOL_DESCRIPTIONS = {
62
+ chat_completion: 'Send messages to an OpenRouter model and get a text response. Supports provider routing ' +
63
+ '(quantizations / ignore / sort / order / require_parameters / data_collection / allow_fallbacks), ' +
64
+ 'model variant suffixes (`:nitro` fastest, `:floor` cheapest, `:exacto` tool-calling accuracy), ' +
65
+ 'reasoning token passthrough, web search, and response caching.\n\n' +
66
+ 'Fails when:\n' +
67
+ '- INVALID_INPUT: messages array is empty\n' +
68
+ '- UPSTREAM_REFUSED: provider rejected the request (credits, content policy, or rate limit)\n' +
69
+ '- UPSTREAM_TIMEOUT: upstream did not respond within the SDK timeout\n' +
70
+ '- MODEL_NOT_FOUND: model slug does not exist on OpenRouter\n\n' +
71
+ 'Works with: validate_model (pre-flight model id check), search_models (discover models).',
72
+ analyze_image: 'Analyze an image with a vision model. Accepts local file paths, http(s) URLs, or base64 data URLs. ' +
73
+ 'Output is model-generated and tagged `_meta.content_is_untrusted: true`.\n\n' +
74
+ 'Fails when:\n' +
75
+ '- INVALID_INPUT: image_path missing or malformed\n' +
76
+ '- UNSAFE_PATH: local path escaped the sandbox\n' +
77
+ '- RESOURCE_TOO_LARGE: image exceeded configured fetch size cap\n' +
78
+ '- UPSTREAM_REFUSED: provider SSRF guard blocked the URL or content policy rejected\n\n' +
79
+ 'Works with: search_models (find vision-capable models), generate_image (follow-up creation).',
80
+ analyze_audio: 'Transcribe or analyze an audio file (WAV / MP3 / FLAC / OGG / etc.) using a multimodal model. ' +
81
+ 'Output is tagged `_meta.content_is_untrusted: true`.\n\n' +
82
+ 'Fails when:\n' +
83
+ '- INVALID_INPUT: audio_path missing\n' +
84
+ '- UNSUPPORTED_FORMAT: decoder could not identify the file as audio\n' +
85
+ '- RESOURCE_TOO_LARGE: input exceeded size cap\n' +
86
+ '- UPSTREAM_REFUSED: blocked host or content policy\n\n' +
87
+ 'Works with: generate_audio (text-to-speech follow-up).',
88
+ analyze_video: 'Describe or analyze a video (mp4 / mpeg / mov / webm) using a multimodal model. Default model: ' +
89
+ 'google/gemini-2.5-flash. Output is tagged `_meta.content_is_untrusted: true`.\n\n' +
90
+ 'Fails when:\n' +
91
+ '- INVALID_INPUT: video_path missing\n' +
92
+ '- UNSUPPORTED_FORMAT: not a recognized video container\n' +
93
+ '- RESOURCE_TOO_LARGE: exceeds fetch cap\n' +
94
+ '- UPSTREAM_REFUSED: SSRF block or provider refusal\n\n' +
95
+ 'Works with: generate_video (text-to-video), get_video_status (poll async jobs).',
96
+ search_models: 'Search OpenRouter\'s model catalog by name, provider, or capability. Returns a paginated list; ' +
97
+ 'use `offset` / `limit` / `next_offset` to page through.\n\n' +
98
+ 'Fails when:\n' +
99
+ '- UPSTREAM_HTTP: /models endpoint returned an error\n' +
100
+ '- UPSTREAM_REFUSED: invalid API key\n\n' +
101
+ 'Works with: validate_model, get_model_info.',
102
+ get_model_info: 'Get pricing / context-length / capability details for a specific model id.\n\n' +
103
+ 'Fails when:\n' +
104
+ '- INVALID_INPUT: model not provided\n' +
105
+ '- MODEL_NOT_FOUND: model slug does not exist\n' +
106
+ '- UPSTREAM_HTTP: model list fetch failed\n\n' +
107
+ 'Works with: search_models (discover ids), validate_model (cheap existence check).',
108
+ validate_model: 'Check whether a model id exists on OpenRouter. Cheap boolean lookup against the cached catalog.\n\n' +
109
+ 'Fails when:\n' +
110
+ '- INVALID_INPUT: model not provided\n' +
111
+ '- UPSTREAM_HTTP: catalog refresh failed\n\n' +
112
+ 'Works with: get_model_info (detailed lookup), chat_completion (pre-flight validation).',
113
+ generate_image: 'Generate an image from a text prompt. Optional reference images condition the output for style / ' +
114
+ 'identity consistency. Default model: google/gemini-2.5-flash-image.\n\n' +
115
+ 'Fails when:\n' +
116
+ '- INVALID_INPUT: prompt empty, bad aspect_ratio / image_size, unreadable reference image\n' +
117
+ '- UNSAFE_PATH: save_path or input_images path escaped the sandbox\n' +
118
+ '- UPSTREAM_REFUSED: provider content policy rejected or insufficient credits\n' +
119
+ '- MODEL_NOT_FOUND: model slug invalid\n\n' +
120
+ 'Works with: analyze_image (verify the result), generate_video_from_image (next step in workflow).',
121
+ generate_audio: 'Generate audio (speech / music) from a text prompt. Format auto-detected, extension auto-corrected.\n\n' +
122
+ 'Fails when:\n' +
123
+ '- INVALID_INPUT: prompt empty\n' +
124
+ '- UNSAFE_PATH: save_path escaped the sandbox\n' +
125
+ '- UPSTREAM_REFUSED: content policy or credit issues\n\n' +
126
+ 'Works with: analyze_audio (verify the result).',
127
+ generate_video: 'Generate a video from a text prompt (optionally conditioned on first/last-frame or reference images). ' +
128
+ 'Submits an async job, polls until completion or max_wait_ms, and downloads the result. Emits MCP ' +
129
+ 'progress notifications when the client provides a `progressToken`. Default model: google/veo-3.1.\n\n' +
130
+ 'Fails when:\n' +
131
+ '- INVALID_INPUT: prompt empty\n' +
132
+ '- UNSAFE_PATH: save_path or reference image paths escaped the sandbox\n' +
133
+ '- UPSTREAM_REFUSED: content policy, credits, or bad request\n' +
134
+ '- JOB_FAILED: provider marked the job as failed\n' +
135
+ '- UNSUPPORTED_FORMAT: reference/frame image could not be decoded\n\n' +
136
+ 'Returns successfully with `_meta.code: JOB_STILL_RUNNING` (NOT an error) when the timeout ' +
137
+ 'elapses — the response carries `_meta.video_id` so callers can resume via get_video_status.\n\n' +
138
+ 'Works with: get_video_status (resume timed-out jobs), generate_video_from_image (narrower image-to-video variant).',
139
+ generate_video_from_image: 'Narrower convenience wrapper around generate_video for image-to-video workflows. Takes a single ' +
140
+ '`image` argument (used as the first frame) and `prompt`. Per arxiv 2511.03497, narrower tools with ' +
141
+ 'fewer parameters improve tool-call hit rate. For last-frame conditioning or reference images, use ' +
142
+ 'generate_video directly.\n\n' +
143
+ 'Fails when:\n' +
144
+ '- INVALID_INPUT: image or prompt missing\n' +
145
+ '- UNSAFE_PATH: image path escaped the sandbox\n' +
146
+ '- UPSTREAM_REFUSED / JOB_FAILED: same as generate_video\n\n' +
147
+ 'Returns successfully with `_meta.code: JOB_STILL_RUNNING` on timeout (resumable via ' +
148
+ 'get_video_status).\n\n' +
149
+ 'Works with: generate_video (full parameter surface), get_video_status.',
150
+ get_video_status: 'Poll an async video-generation job by id. Downloads the result when complete (and saves if save_path given).\n\n' +
151
+ 'Fails when:\n' +
152
+ '- INVALID_INPUT: video_id missing\n' +
153
+ '- UNSAFE_PATH: save_path escaped the sandbox\n' +
154
+ '- JOB_FAILED: provider marked the job as failed\n\n' +
155
+ 'Returns successfully with `_meta.code: JOB_STILL_RUNNING` (NOT an error) when the job is still ' +
156
+ 'in flight — response carries `_meta.last_status` and `_meta.progress` so callers can retry later.\n\n' +
157
+ 'Works with: generate_video, generate_video_from_image.',
158
+ rerank_documents: 'Re-order a list of documents by relevance to a query using an OpenRouter reranker. Default model: ' +
159
+ 'cohere/rerank-english-v3.0.\n\n' +
160
+ 'Fails when:\n' +
161
+ '- INVALID_INPUT: query missing, documents empty, non-string document elements\n' +
162
+ '- MODEL_NOT_FOUND: reranker model slug does not exist\n' +
163
+ '- UPSTREAM_HTTP: provider returned an error\n\n' +
164
+ 'Works with: search_models (discover rerankers), chat_completion (answer grounded in top-ranked docs).',
165
+ health_check: 'Verify API-key validity, OpenRouter reachability, and return server + protocol versions. No args.\n\n' +
166
+ 'Fails when: never returns an error result — always returns `{ ok, api_key_valid, ... }` so ops can ' +
167
+ 'programmatically branch on the payload.\n\n' +
168
+ 'Works with: every other tool (run once at startup to confirm credentials).',
169
+ };
18
170
  export class ToolHandlers {
19
171
  openai;
20
172
  modelCache = ModelCache.getInstance();
21
173
  apiClient;
22
174
  defaultModel;
175
+ server;
23
176
  constructor(server, apiKey, defaultModel) {
24
177
  this.defaultModel = defaultModel;
25
178
  this.apiClient = new OpenRouterAPIClient(apiKey);
@@ -27,6 +180,7 @@ export class ToolHandlers {
27
180
  apiKey,
28
181
  baseURL: 'https://openrouter.ai/api/v1',
29
182
  });
183
+ this.server = server;
30
184
  this.register(server);
31
185
  }
32
186
  register(server) {
@@ -34,18 +188,22 @@ export class ToolHandlers {
34
188
  tools: [
35
189
  {
36
190
  name: 'chat_completion',
37
- description: 'Send messages to an OpenRouter model and get a response. Supports provider routing (quantizations / ignore / sort / order / require_parameters / data_collection / allow_fallbacks) and model variant suffixes (`:nitro` for faster, `:floor` for cheapest).',
191
+ description: TOOL_DESCRIPTIONS.chat_completion,
38
192
  annotations: {
193
+ title: 'Chat completion',
39
194
  readOnlyHint: false,
40
195
  destructiveHint: false,
41
196
  idempotentHint: false,
197
+ openWorldHint: true,
42
198
  },
43
199
  inputSchema: {
44
200
  type: 'object',
45
201
  properties: {
46
202
  model: {
47
203
  type: 'string',
48
- description: 'Model ID (optional, uses default). Append `:nitro` for faster/experimental variants or `:floor` for the cheapest available variant (e.g. `openai/gpt-4o:nitro`).',
204
+ description: 'Model ID (optional, uses default). Append `:nitro` for the fastest variant, ' +
205
+ '`:floor` for the cheapest, or `:exacto` for the best tool-calling accuracy. ' +
206
+ 'Example: `openai/gpt-4o:nitro`.',
49
207
  },
50
208
  messages: {
51
209
  type: 'array',
@@ -69,7 +227,8 @@ export class ToolHandlers {
69
227
  },
70
228
  provider: {
71
229
  type: 'object',
72
- description: 'OpenRouter provider-routing overrides. Merges on top of `OPENROUTER_PROVIDER_*` env defaults. See https://openrouter.ai/docs/features/provider-routing',
230
+ description: 'OpenRouter provider-routing overrides. Merges on top of `OPENROUTER_PROVIDER_*` env defaults. ' +
231
+ 'See https://openrouter.ai/docs/features/provider-routing',
73
232
  properties: {
74
233
  quantizations: {
75
234
  type: 'array',
@@ -79,44 +238,57 @@ export class ToolHandlers {
79
238
  ignore: {
80
239
  type: 'array',
81
240
  items: { type: 'string' },
82
- description: 'Exclude these provider slugs (e.g. `["openai","anthropic"]`).',
241
+ description: 'Exclude these provider slugs.',
83
242
  },
84
243
  sort: {
85
244
  type: 'string',
86
245
  enum: ['price', 'throughput', 'latency'],
87
- description: 'Sort providers by this criterion.',
88
- },
89
- order: {
90
- type: 'array',
91
- items: { type: 'string' },
92
- description: 'Prioritized list of provider IDs (e.g. `["openai/gpt-4o","anthropic/claude-3-opus"]`).',
93
- },
94
- require_parameters: {
95
- type: 'boolean',
96
- description: 'Only use providers that support every parameter in the request.',
97
- },
98
- data_collection: {
99
- type: 'string',
100
- enum: ['allow', 'deny'],
101
- description: 'Whether providers may collect request data.',
102
- },
103
- allow_fallbacks: {
104
- type: 'boolean',
105
- description: 'Allow fallback to unlisted providers when preferred ones fail.',
106
246
  },
247
+ order: { type: 'array', items: { type: 'string' } },
248
+ require_parameters: { type: 'boolean' },
249
+ data_collection: { type: 'string', enum: ['allow', 'deny'] },
250
+ allow_fallbacks: { type: 'boolean' },
107
251
  },
108
252
  },
253
+ include_reasoning: {
254
+ type: 'boolean',
255
+ description: 'Surface the model\'s chain-of-thought on `_meta.reasoning` for R1 / Opus 4.7 / Gemini Thinking.',
256
+ },
257
+ online: {
258
+ type: 'boolean',
259
+ description: 'Enable OpenRouter\'s web-search plugin (Exa-backed, $4 / 1000 results).',
260
+ },
261
+ web_max_results: {
262
+ type: 'number',
263
+ minimum: 1,
264
+ description: 'Max web-search results when `online: true` (default 5).',
265
+ },
266
+ cache: {
267
+ type: 'boolean',
268
+ description: 'Enable OpenRouter response caching via `X-OpenRouter-Cache: true`. ' +
269
+ 'Server-wide default settable via `OPENROUTER_CACHE_RESPONSES=1`.',
270
+ },
271
+ cache_ttl: {
272
+ type: 'string',
273
+ description: 'Cache TTL (e.g. `"5m"`, `"1h"`, `"24h"`; 1s-24h range).',
274
+ },
275
+ cache_clear: {
276
+ type: 'boolean',
277
+ description: 'Bust the cache entry for this exact request.',
278
+ },
109
279
  },
110
280
  required: ['messages'],
111
281
  },
112
282
  },
113
283
  {
114
284
  name: 'analyze_image',
115
- description: 'Analyze an image using a vision model',
285
+ description: TOOL_DESCRIPTIONS.analyze_image,
116
286
  annotations: {
287
+ title: 'Analyze image',
117
288
  readOnlyHint: true,
118
289
  destructiveHint: false,
119
290
  idempotentHint: false,
291
+ openWorldHint: true,
120
292
  },
121
293
  inputSchema: {
122
294
  type: 'object',
@@ -124,17 +296,27 @@ export class ToolHandlers {
124
296
  image_path: { type: 'string', description: 'File path, URL, or data URL' },
125
297
  question: { type: 'string', description: 'Question about the image' },
126
298
  model: { type: 'string' },
299
+ cache_input: {
300
+ type: 'boolean',
301
+ description: 'Attach `cache_control: ephemeral` to the image block so Anthropic / Gemini prompt-cache it. ' +
302
+ 'Repeat questions about the same image save ~10x on Anthropic.',
303
+ },
304
+ cache: { type: 'boolean' },
305
+ cache_ttl: { type: 'string' },
306
+ cache_clear: { type: 'boolean' },
127
307
  },
128
308
  required: ['image_path'],
129
309
  },
130
310
  },
131
311
  {
132
312
  name: 'analyze_audio',
133
- description: 'Analyze or transcribe an audio file using a multimodal model',
313
+ description: TOOL_DESCRIPTIONS.analyze_audio,
134
314
  annotations: {
315
+ title: 'Analyze audio',
135
316
  readOnlyHint: true,
136
317
  destructiveHint: false,
137
318
  idempotentHint: false,
319
+ openWorldHint: true,
138
320
  },
139
321
  inputSchema: {
140
322
  type: 'object',
@@ -148,41 +330,50 @@ export class ToolHandlers {
148
330
  description: 'Question or instruction about the audio (default: transcribe)',
149
331
  },
150
332
  model: { type: 'string' },
333
+ cache_input: { type: 'boolean' },
334
+ cache: { type: 'boolean' },
335
+ cache_ttl: { type: 'string' },
336
+ cache_clear: { type: 'boolean' },
151
337
  },
152
338
  required: ['audio_path'],
153
339
  },
154
340
  },
155
341
  {
156
342
  name: 'analyze_video',
157
- description: 'Analyze or transcribe a video file using a multimodal model. Accepts mp4, mpeg, mov, or webm from a local file path, HTTP(S) URL, or base64 data URL. Default model: google/gemini-2.5-flash.',
343
+ description: TOOL_DESCRIPTIONS.analyze_video,
158
344
  annotations: {
345
+ title: 'Analyze video',
159
346
  readOnlyHint: true,
160
347
  destructiveHint: false,
161
348
  idempotentHint: false,
349
+ openWorldHint: true,
162
350
  },
163
351
  inputSchema: {
164
352
  type: 'object',
165
353
  properties: {
166
354
  video_path: {
167
355
  type: 'string',
168
- description: 'File path, HTTP(S) URL, or base64 data URL. Supported formats: mp4, mpeg, mov, webm.',
356
+ description: 'File path, HTTP(S) URL, or base64 data URL. Supported: mp4 / mpeg / mov / webm.',
169
357
  },
170
- question: {
171
- type: 'string',
172
- description: 'Question or instruction about the video (default: describe).',
173
- },
174
- model: { type: 'string', description: 'Override the model ID.' },
358
+ question: { type: 'string' },
359
+ model: { type: 'string' },
360
+ cache_input: { type: 'boolean' },
361
+ cache: { type: 'boolean' },
362
+ cache_ttl: { type: 'string' },
363
+ cache_clear: { type: 'boolean' },
175
364
  },
176
365
  required: ['video_path'],
177
366
  },
178
367
  },
179
368
  {
180
369
  name: 'search_models',
181
- description: 'Search available OpenRouter models',
370
+ description: TOOL_DESCRIPTIONS.search_models,
182
371
  annotations: {
372
+ title: 'Search models',
183
373
  readOnlyHint: true,
184
374
  destructiveHint: false,
185
375
  idempotentHint: true,
376
+ openWorldHint: true,
186
377
  },
187
378
  inputSchema: {
188
379
  type: 'object',
@@ -198,47 +389,81 @@ export class ToolHandlers {
198
389
  },
199
390
  },
200
391
  limit: { type: 'number', minimum: 1, maximum: 50 },
392
+ offset: { type: 'number', minimum: 0 },
201
393
  },
202
394
  },
395
+ outputSchema: {
396
+ type: 'object',
397
+ properties: {
398
+ results: { type: 'array', items: { type: 'object' } },
399
+ offset: { type: 'number' },
400
+ limit: { type: 'number' },
401
+ total: { type: 'number' },
402
+ has_more: { type: 'boolean' },
403
+ next_offset: { type: ['number', 'null'] },
404
+ },
405
+ required: ['results', 'offset', 'limit', 'total', 'has_more', 'next_offset'],
406
+ },
203
407
  },
204
408
  {
205
409
  name: 'get_model_info',
206
- description: 'Get details about a specific model',
410
+ description: TOOL_DESCRIPTIONS.get_model_info,
207
411
  annotations: {
412
+ title: 'Get model info',
208
413
  readOnlyHint: true,
209
414
  destructiveHint: false,
210
415
  idempotentHint: true,
416
+ openWorldHint: true,
211
417
  },
212
418
  inputSchema: {
213
419
  type: 'object',
214
420
  properties: { model: { type: 'string' } },
215
421
  required: ['model'],
216
422
  },
423
+ outputSchema: {
424
+ type: 'object',
425
+ properties: {
426
+ id: { type: 'string' },
427
+ name: { type: 'string' },
428
+ context_length: { type: 'number' },
429
+ architecture: { type: 'object' },
430
+ },
431
+ required: ['id'],
432
+ },
217
433
  },
218
434
  {
219
435
  name: 'validate_model',
220
- description: 'Check if a model ID exists',
436
+ description: TOOL_DESCRIPTIONS.validate_model,
221
437
  annotations: {
438
+ title: 'Validate model',
222
439
  readOnlyHint: true,
223
440
  destructiveHint: false,
224
441
  idempotentHint: true,
442
+ openWorldHint: true,
225
443
  },
226
444
  inputSchema: {
227
445
  type: 'object',
228
446
  properties: { model: { type: 'string' } },
229
447
  required: ['model'],
230
448
  },
449
+ outputSchema: {
450
+ type: 'object',
451
+ properties: {
452
+ valid: { type: 'boolean' },
453
+ model: { type: 'string' },
454
+ },
455
+ required: ['valid', 'model'],
456
+ },
231
457
  },
232
458
  {
233
459
  name: 'generate_image',
234
- description: 'Generate an image from a text prompt. Optionally conditioned on one or more ' +
235
- 'reference images (file paths, http(s) URLs, or data URLs) for character / style ' +
236
- 'consistency. Sends `modalities: ["image","text"]` by default; override via the ' +
237
- '`modalities` field if needed.',
460
+ description: TOOL_DESCRIPTIONS.generate_image,
238
461
  annotations: {
462
+ title: 'Generate image',
239
463
  readOnlyHint: false,
240
464
  destructiveHint: false,
241
465
  idempotentHint: false,
466
+ openWorldHint: true,
242
467
  },
243
468
  inputSchema: {
244
469
  type: 'object',
@@ -247,7 +472,6 @@ export class ToolHandlers {
247
472
  model: { type: 'string' },
248
473
  aspect_ratio: {
249
474
  type: 'string',
250
- description: 'Output aspect ratio (e.g. 1:1, 16:9, 9:16, 4:3, 3:4, 21:9). Model-dependent.',
251
475
  enum: [
252
476
  '1:1',
253
477
  '2:3',
@@ -265,181 +489,227 @@ export class ToolHandlers {
265
489
  '8:1',
266
490
  ],
267
491
  },
268
- image_size: {
269
- type: 'string',
270
- description: 'Output resolution bucket. 1K is the default; 0.5K / 2K / 4K are model-dependent.',
271
- enum: ['0.5K', '1K', '2K', '4K'],
272
- },
273
- max_tokens: {
274
- type: 'number',
275
- minimum: 1,
276
- description: 'Cap on completion tokens. Defaults to the model context window, which can trip free-tier quotas; set e.g. 4096 on low-credit accounts.',
277
- },
278
- save_path: {
279
- type: 'string',
280
- description: 'Optional path to save the image. Routed through the OPENROUTER_OUTPUT_DIR sandbox.',
281
- },
282
- input_images: {
283
- type: 'array',
284
- items: { type: 'string' },
285
- description: 'Optional reference images for visual consistency. Each entry may be a ' +
286
- 'local file path (sandboxed to OPENROUTER_INPUT_DIR / OPENROUTER_OUTPUT_DIR / ' +
287
- 'cwd), an http(s) URL, or a `data:image/...;base64,...` URL. Inlined as ' +
288
- 'multimodal user content in the order given.',
289
- },
290
- modalities: {
291
- type: 'array',
292
- items: { type: 'string' },
293
- description: 'Override the default `modalities: ["image","text"]` sent to OpenRouter. ' +
294
- 'Most callers should leave this unset. Provide e.g. ["text"] to suppress ' +
295
- 'image output for inspection / captioning.',
296
- },
492
+ image_size: { type: 'string', enum: ['0.5K', '1K', '2K', '4K'] },
493
+ max_tokens: { type: 'number', minimum: 1 },
494
+ save_path: { type: 'string' },
495
+ input_images: { type: 'array', items: { type: 'string' } },
496
+ modalities: { type: 'array', items: { type: 'string' } },
297
497
  },
298
498
  required: ['prompt'],
299
499
  },
300
500
  },
301
501
  {
302
502
  name: 'generate_audio',
303
- description: 'Generate audio from a text prompt. Conversational models (e.g. openai/gpt-audio) respond in spoken audio. Music models (e.g. google/lyria-3-clip-preview) need a structured prompt. Output format is auto-detected and file extension is corrected automatically.',
503
+ description: TOOL_DESCRIPTIONS.generate_audio,
304
504
  annotations: {
505
+ title: 'Generate audio',
305
506
  readOnlyHint: false,
306
507
  destructiveHint: false,
307
508
  idempotentHint: false,
509
+ openWorldHint: true,
308
510
  },
309
511
  inputSchema: {
310
512
  type: 'object',
311
513
  properties: {
312
- prompt: { type: 'string', description: 'Text input' },
313
- model: { type: 'string', description: 'Model ID (default: openai/gpt-audio)' },
314
- voice: { type: 'string', description: 'Voice name (default: alloy)' },
315
- format: {
316
- type: 'string',
317
- description: 'Requested format: pcm16 (default), mp3, flac, opus',
318
- },
319
- save_path: {
320
- type: 'string',
321
- description: 'Optional path to save the audio. Extension auto-corrected and routed through OPENROUTER_OUTPUT_DIR sandbox.',
322
- },
514
+ prompt: { type: 'string' },
515
+ model: { type: 'string' },
516
+ voice: { type: 'string' },
517
+ format: { type: 'string' },
518
+ save_path: { type: 'string' },
323
519
  },
324
520
  required: ['prompt'],
325
521
  },
326
522
  },
327
523
  {
328
524
  name: 'generate_video',
329
- description: 'Generate a video from a text prompt using an OpenRouter video-generation model (default: google/veo-3.1). ' +
330
- 'Submits an async job, polls until completion or max_wait_ms, then downloads the result. ' +
331
- 'Optionally conditioned on first/last-frame images or reference images. ' +
332
- 'Large outputs are auto-saved when save_path is provided and path-sandboxed.',
525
+ description: TOOL_DESCRIPTIONS.generate_video,
333
526
  annotations: {
527
+ title: 'Generate video',
334
528
  readOnlyHint: false,
335
529
  destructiveHint: false,
336
530
  idempotentHint: false,
531
+ openWorldHint: true,
337
532
  },
338
533
  inputSchema: {
339
534
  type: 'object',
340
535
  properties: {
341
- prompt: { type: 'string', description: 'Text description of the desired video.' },
342
- model: { type: 'string', description: 'Override the video model ID.' },
343
- resolution: {
344
- type: 'string',
345
- description: '480p / 720p / 1080p / 1K / 2K / 4K (model-dependent).',
346
- },
347
- aspect_ratio: {
348
- type: 'string',
349
- description: '16:9 / 9:16 / 1:1 / 4:3 / 3:4 / 21:9 / 9:21 (model-dependent).',
350
- },
351
- duration: {
352
- type: 'number',
353
- minimum: 1,
354
- description: 'Duration in seconds (model-dependent).',
355
- },
356
- seed: { type: 'number', description: 'Deterministic seed when supported.' },
357
- first_frame_image: {
358
- type: 'string',
359
- description: 'Optional image (path, URL, or data URL) used as the first frame for image-to-video.',
360
- },
361
- last_frame_image: {
362
- type: 'string',
363
- description: 'Optional image used as the last frame for frame transitions.',
364
- },
365
- reference_images: {
366
- type: 'array',
367
- items: { type: 'string' },
368
- description: 'Optional style/content reference images.',
369
- },
370
- provider: {
371
- type: 'object',
372
- description: 'Provider-specific passthrough options keyed by provider slug.',
373
- },
374
- save_path: {
536
+ prompt: { type: 'string' },
537
+ model: { type: 'string' },
538
+ resolution: { type: 'string' },
539
+ aspect_ratio: { type: 'string' },
540
+ duration: { type: 'number', minimum: 1 },
541
+ seed: { type: 'number' },
542
+ first_frame_image: { type: 'string' },
543
+ last_frame_image: { type: 'string' },
544
+ reference_images: { type: 'array', items: { type: 'string' } },
545
+ provider: { type: 'object' },
546
+ save_path: { type: 'string' },
547
+ max_wait_ms: { type: 'number', minimum: 10000 },
548
+ poll_interval_ms: { type: 'number', minimum: 2000 },
549
+ },
550
+ required: ['prompt'],
551
+ },
552
+ },
553
+ {
554
+ name: 'generate_video_from_image',
555
+ description: TOOL_DESCRIPTIONS.generate_video_from_image,
556
+ annotations: {
557
+ title: 'Generate video from image',
558
+ readOnlyHint: false,
559
+ destructiveHint: false,
560
+ idempotentHint: false,
561
+ openWorldHint: true,
562
+ },
563
+ inputSchema: {
564
+ type: 'object',
565
+ properties: {
566
+ image: {
375
567
  type: 'string',
376
- description: 'Where to save the video. Routed through the OPENROUTER_OUTPUT_DIR sandbox; extension auto-corrected.',
377
- },
378
- max_wait_ms: {
379
- type: 'number',
380
- minimum: 10000,
381
- description: 'Total time to wait for the async job before returning a resumable handle (default 600000 ms).',
382
- },
383
- poll_interval_ms: {
384
- type: 'number',
385
- minimum: 2000,
386
- description: 'Polling cadence (default 15000 ms).',
568
+ description: 'First-frame image (path, URL, or data URL). Required.',
387
569
  },
570
+ prompt: { type: 'string' },
571
+ model: { type: 'string' },
572
+ resolution: { type: 'string' },
573
+ aspect_ratio: { type: 'string' },
574
+ duration: { type: 'number', minimum: 1 },
575
+ seed: { type: 'number' },
576
+ save_path: { type: 'string' },
577
+ max_wait_ms: { type: 'number', minimum: 10000 },
578
+ poll_interval_ms: { type: 'number', minimum: 2000 },
388
579
  },
389
- required: ['prompt'],
580
+ required: ['image', 'prompt'],
390
581
  },
391
582
  },
392
583
  {
393
584
  name: 'get_video_status',
394
- description: 'Resume a previously submitted video generation job by id. Returns the latest status; if completed, ' +
395
- 'downloads the video (and saves it when save_path is provided).',
585
+ description: TOOL_DESCRIPTIONS.get_video_status,
396
586
  annotations: {
587
+ title: 'Get video status',
397
588
  readOnlyHint: true,
398
589
  destructiveHint: false,
399
590
  idempotentHint: true,
591
+ openWorldHint: true,
400
592
  },
401
593
  inputSchema: {
402
594
  type: 'object',
403
595
  properties: {
404
- video_id: { type: 'string', description: 'Job id from a previous generate_video call.' },
405
- save_path: {
406
- type: 'string',
407
- description: 'Optional save path (applies when the job is already completed).',
408
- },
596
+ video_id: { type: 'string' },
597
+ save_path: { type: 'string' },
409
598
  },
410
599
  required: ['video_id'],
411
600
  },
412
601
  },
602
+ {
603
+ name: 'rerank_documents',
604
+ description: TOOL_DESCRIPTIONS.rerank_documents,
605
+ annotations: {
606
+ title: 'Rerank documents',
607
+ readOnlyHint: true,
608
+ destructiveHint: false,
609
+ idempotentHint: true,
610
+ openWorldHint: true,
611
+ },
612
+ inputSchema: {
613
+ type: 'object',
614
+ properties: {
615
+ query: { type: 'string' },
616
+ documents: { type: 'array', items: { type: 'string' }, minItems: 1 },
617
+ model: { type: 'string' },
618
+ top_n: { type: 'number', minimum: 1 },
619
+ return_documents: { type: 'boolean' },
620
+ },
621
+ required: ['query', 'documents'],
622
+ },
623
+ outputSchema: {
624
+ type: 'object',
625
+ properties: {
626
+ model: { type: 'string' },
627
+ results: {
628
+ type: 'array',
629
+ items: {
630
+ type: 'object',
631
+ properties: {
632
+ index: { type: 'number' },
633
+ score: { type: 'number' },
634
+ document: { type: 'string' },
635
+ },
636
+ required: ['index', 'score'],
637
+ },
638
+ },
639
+ },
640
+ required: ['results'],
641
+ },
642
+ },
643
+ {
644
+ name: 'health_check',
645
+ description: TOOL_DESCRIPTIONS.health_check,
646
+ annotations: {
647
+ title: 'Health check',
648
+ readOnlyHint: true,
649
+ destructiveHint: false,
650
+ idempotentHint: true,
651
+ openWorldHint: true,
652
+ },
653
+ inputSchema: { type: 'object', properties: {} },
654
+ outputSchema: {
655
+ type: 'object',
656
+ properties: {
657
+ ok: { type: 'boolean' },
658
+ server_version: { type: 'string' },
659
+ protocol_version: { type: 'string' },
660
+ api_key_valid: { type: 'boolean' },
661
+ models_cached: { type: 'number' },
662
+ error: { type: 'string' },
663
+ },
664
+ required: ['ok', 'server_version', 'protocol_version', 'api_key_valid', 'models_cached'],
665
+ },
666
+ },
413
667
  ],
414
668
  }));
415
669
  server.setRequestHandler(CallToolRequestSchema, async (request) => {
416
670
  const { name, arguments: args } = request.params;
417
- switch (name) {
418
- case 'chat_completion':
419
- return handleChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
420
- case 'analyze_image':
421
- return handleAnalyzeImage(wrapToolArgs(args), this.openai, this.defaultModel);
422
- case 'analyze_audio':
423
- return handleAnalyzeAudio(wrapToolArgs(args), this.openai, this.defaultModel);
424
- case 'analyze_video':
425
- return handleAnalyzeVideo(wrapToolArgs(args), this.openai, this.defaultModel);
426
- case 'search_models':
427
- return handleSearchModels(wrapToolArgs(args), this.apiClient, this.modelCache);
428
- case 'get_model_info':
429
- return handleGetModelInfo(wrapToolArgs(args), this.modelCache, this.apiClient);
430
- case 'validate_model':
431
- return handleValidateModel(wrapToolArgs(args), this.modelCache, this.apiClient);
432
- case 'generate_image':
433
- return handleGenerateImage(wrapToolArgs(args), this.openai);
434
- case 'generate_audio':
435
- return handleGenerateAudio(wrapToolArgs(args), this.openai);
436
- case 'generate_video':
437
- return handleGenerateVideo(wrapToolArgs(args), this.apiClient);
438
- case 'get_video_status':
439
- return handleGetVideoStatus(wrapToolArgs(args), this.apiClient);
440
- default:
441
- throw new McpError(McpErrorCode.MethodNotFound, `Unknown tool: ${name}`);
442
- }
671
+ // Our handlers return structured shapes that satisfy CallToolResult's
672
+ // open interface (content + optional _meta + optional isError +
673
+ // optional structuredContent). We cast through unknown because
674
+ // several handlers include server-specific _meta keys (e.g.
675
+ // server_version, cache, code) that aren't listed in the SDK's
676
+ // typed schema — the SDK accepts any extras thanks to the
677
+ // `[x: string]: unknown` index signature.
678
+ const dispatch = async () => {
679
+ switch (name) {
680
+ case 'chat_completion':
681
+ return handleChatCompletion(wrapToolArgs(args), this.openai, this.defaultModel);
682
+ case 'analyze_image':
683
+ return handleAnalyzeImage(wrapToolArgs(args), this.openai, this.defaultModel);
684
+ case 'analyze_audio':
685
+ return handleAnalyzeAudio(wrapToolArgs(args), this.openai, this.defaultModel);
686
+ case 'analyze_video':
687
+ return handleAnalyzeVideo(wrapToolArgs(args), this.openai, this.defaultModel);
688
+ case 'search_models':
689
+ return handleSearchModels(wrapToolArgs(args), this.apiClient, this.modelCache);
690
+ case 'get_model_info':
691
+ return handleGetModelInfo(wrapToolArgs(args), this.modelCache, this.apiClient);
692
+ case 'validate_model':
693
+ return handleValidateModel(wrapToolArgs(args), this.modelCache, this.apiClient);
694
+ case 'generate_image':
695
+ return handleGenerateImage(wrapToolArgs(args), this.openai);
696
+ case 'generate_audio':
697
+ return handleGenerateAudio(wrapToolArgs(args), this.openai);
698
+ case 'generate_video':
699
+ return handleGenerateVideo(wrapToolArgs(args), this.apiClient, buildProgressHook(this.server, extractProgressToken(request)));
700
+ case 'generate_video_from_image':
701
+ return handleGenerateVideoFromImage(wrapToolArgs(args), this.apiClient, buildProgressHook(this.server, extractProgressToken(request)));
702
+ case 'get_video_status':
703
+ return handleGetVideoStatus(wrapToolArgs(args), this.apiClient);
704
+ case 'rerank_documents':
705
+ return handleRerankDocuments(wrapToolArgs(args), this.apiClient);
706
+ case 'health_check':
707
+ return handleHealthCheck(wrapToolArgs(args), this.apiClient, this.modelCache);
708
+ default:
709
+ throw new McpError(McpErrorCode.MethodNotFound, `Unknown tool: ${name}`);
710
+ }
711
+ };
712
+ return (await dispatch());
443
713
  });
444
714
  }
445
715
  }