@stabgan/openrouter-mcp-multimodal 4.7.0 → 4.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/README.md +90 -36
  2. package/dist/errors.d.ts +5 -20
  3. package/dist/errors.js +1 -10
  4. package/dist/index.js +7 -1
  5. package/dist/logger.js +54 -24
  6. package/dist/model-cache.d.ts +13 -0
  7. package/dist/model-cache.js +61 -5
  8. package/dist/openrouter-api.d.ts +14 -15
  9. package/dist/openrouter-api.js +68 -22
  10. package/dist/tool-definitions.d.ts +24 -0
  11. package/dist/tool-definitions.js +276 -174
  12. package/dist/tool-descriptions.js +23 -15
  13. package/dist/tool-handlers/analyze-audio.js +4 -1
  14. package/dist/tool-handlers/analyze-image.js +10 -5
  15. package/dist/tool-handlers/analyze-video.js +9 -5
  16. package/dist/tool-handlers/async-chat.d.ts +17 -0
  17. package/dist/tool-handlers/async-chat.js +104 -30
  18. package/dist/tool-handlers/audio-utils.d.ts +19 -4
  19. package/dist/tool-handlers/audio-utils.js +170 -16
  20. package/dist/tool-handlers/cache.d.ts +3 -3
  21. package/dist/tool-handlers/cache.js +56 -4
  22. package/dist/tool-handlers/chat-completion.js +16 -7
  23. package/dist/tool-handlers/chat-request.d.ts +3 -0
  24. package/dist/tool-handlers/chat-request.js +28 -0
  25. package/dist/tool-handlers/completion-utils.d.ts +5 -11
  26. package/dist/tool-handlers/completion-utils.js +76 -47
  27. package/dist/tool-handlers/fetch-utils.d.ts +14 -0
  28. package/dist/tool-handlers/fetch-utils.js +331 -68
  29. package/dist/tool-handlers/generate-audio.d.ts +4 -15
  30. package/dist/tool-handlers/generate-audio.js +21 -53
  31. package/dist/tool-handlers/generate-image-dedicated.d.ts +1 -1
  32. package/dist/tool-handlers/generate-image-dedicated.js +50 -31
  33. package/dist/tool-handlers/generate-image.d.ts +1 -1
  34. package/dist/tool-handlers/generate-image.js +19 -22
  35. package/dist/tool-handlers/generate-video.d.ts +4 -3
  36. package/dist/tool-handlers/generate-video.js +42 -18
  37. package/dist/tool-handlers/get-model-info.js +1 -1
  38. package/dist/tool-handlers/health-check.js +39 -15
  39. package/dist/tool-handlers/image-utils.js +2 -2
  40. package/dist/tool-handlers/openrouter-errors.d.ts +2 -0
  41. package/dist/tool-handlers/openrouter-errors.js +138 -31
  42. package/dist/tool-handlers/path-safety.js +49 -17
  43. package/dist/tool-handlers/path-utils.d.ts +2 -0
  44. package/dist/tool-handlers/path-utils.js +13 -0
  45. package/dist/tool-handlers/provider-routing.d.ts +2 -0
  46. package/dist/tool-handlers/provider-routing.js +10 -0
  47. package/dist/tool-handlers/rerank.d.ts +1 -4
  48. package/dist/tool-handlers/rerank.js +43 -14
  49. package/dist/tool-handlers/search-models.js +3 -3
  50. package/dist/tool-handlers/speech-to-text.d.ts +1 -0
  51. package/dist/tool-handlers/speech-to-text.js +23 -56
  52. package/dist/tool-handlers/text-to-speech.d.ts +1 -1
  53. package/dist/tool-handlers/text-to-speech.js +18 -8
  54. package/dist/tool-handlers/tool-result-payload.js +11 -9
  55. package/dist/tool-handlers/validate-model.js +1 -1
  56. package/dist/tool-handlers.d.ts +9 -0
  57. package/dist/tool-handlers.js +17 -5
  58. package/dist/version.d.ts +1 -1
  59. package/dist/version.js +1 -1
  60. package/package.json +1 -3
@@ -1,8 +1,83 @@
1
- import { TOOL_DESCRIPTIONS } from './tool-descriptions.js';
2
- /** Shared JSON-schema fragment for save_path on generate/write tools. */
3
- const SAVE_PATH_PROPERTY = {
1
+ import { TOOL_DESCRIPTIONS, TOOL_NAMES } from './tool-descriptions.js';
2
+ /** Aspect ratios accepted by generate_image and generate_image_dedicated handlers. */
3
+ export const IMAGE_ASPECT_RATIOS = [
4
+ '1:1',
5
+ '2:3',
6
+ '3:2',
7
+ '3:4',
8
+ '4:3',
9
+ '4:5',
10
+ '5:4',
11
+ '9:16',
12
+ '16:9',
13
+ '21:9',
14
+ '1:4',
15
+ '4:1',
16
+ '1:8',
17
+ '8:1',
18
+ ];
19
+ export const IMAGE_SIZES = ['0.5K', '1K', '2K', '4K'];
20
+ export const IMAGE_DEDICATED_RESOLUTIONS = ['512', '0.5K', '1K', '2K', '4K'];
21
+ export const IMAGE_DEDICATED_QUALITIES = ['auto', 'low', 'medium', 'high'];
22
+ export const IMAGE_OUTPUT_FORMATS = ['png', 'jpeg', 'webp', 'svg'];
23
+ /** generate_audio handler VALID_FORMATS */
24
+ export const GENERATE_AUDIO_FORMATS = ['wav', 'mp3', 'flac', 'opus', 'pcm16'];
25
+ /** text_to_speech handler VALID_FORMATS */
26
+ export const TTS_RESPONSE_FORMATS = ['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm'];
27
+ /** speech_to_text handler VALID_RESPONSE_FORMATS */
28
+ export const STT_RESPONSE_FORMATS = ['json', 'text', 'srt', 'verbose_json', 'vtt'];
29
+ export const CHAT_MESSAGE_ROLES = ['system', 'user', 'assistant'];
30
+ export const PROVIDER_SORT_VALUES = ['price', 'throughput', 'latency'];
31
+ export const PROVIDER_DATA_COLLECTION = ['allow', 'deny'];
32
+ /** Tools whose handlers accept save_path (binary artifact output). */
33
+ export const TOOLS_WITH_SAVE_PATH = [
34
+ 'generate_image',
35
+ 'generate_image_dedicated',
36
+ 'generate_audio',
37
+ 'text_to_speech',
38
+ 'generate_video',
39
+ 'generate_video_from_image',
40
+ 'get_video_status',
41
+ ];
42
+ /** Shared save_path fragment for binary-output tools. */
43
+ export const SAVE_PATH_PROPERTY = {
4
44
  type: 'string',
5
- description: 'Write the artifact under OPENROUTER_OUTPUT_DIR (path-sandboxed). When set, the tool result is text-only with _meta.save_path — no inline media block. Without save_path, inline image/audio/video is returned only if under OPENROUTER_*_INLINE_MAX_BYTES (see .env.example).',
45
+ description: 'Write the artifact under OPENROUTER_OUTPUT_DIR (path-sandboxed). When set, the tool result is text-only with _meta.save_path — no inline media block. ' +
46
+ 'When unset, inline image/audio (default 1 MiB) or video (default 10 MiB) is returned only if under the per-kind ceiling: ' +
47
+ 'OPENROUTER_IMAGE_INLINE_MAX_BYTES, OPENROUTER_AUDIO_INLINE_MAX_BYTES, OPENROUTER_VIDEO_INLINE_MAX_BYTES ' +
48
+ '(global fallback OPENROUTER_INLINE_MAX_BYTES). See .env.example.',
49
+ };
50
+ const SAVE_PATH_WITH_PREFIX = (prefix) => ({
51
+ ...SAVE_PATH_PROPERTY,
52
+ description: `${prefix} ${SAVE_PATH_PROPERTY.description}`,
53
+ });
54
+ const CHAT_MESSAGE_SCHEMA = {
55
+ type: 'array',
56
+ minItems: 1,
57
+ items: {
58
+ type: 'object',
59
+ properties: {
60
+ role: { type: 'string', enum: [...CHAT_MESSAGE_ROLES] },
61
+ content: {
62
+ oneOf: [{ type: 'string' }, { type: 'array', items: { type: 'object' } }],
63
+ },
64
+ },
65
+ required: ['role', 'content'],
66
+ },
67
+ };
68
+ const CACHE_PROPERTIES = {
69
+ cache: {
70
+ type: 'boolean',
71
+ description: 'Enable OpenRouter response caching via `X-OpenRouter-Cache: true`. Server default: `OPENROUTER_CACHE_RESPONSES=1`.',
72
+ },
73
+ cache_ttl: {
74
+ type: 'string',
75
+ description: 'Cache TTL as integer seconds (1–86400) or a duration string such as "30s", "5m", or "1h". Sent upstream as seconds.',
76
+ },
77
+ cache_clear: {
78
+ type: 'boolean',
79
+ description: 'Bust the cache entry for this exact request.',
80
+ },
6
81
  };
7
82
  export const TOOL_DEFINITIONS = [
8
83
  {
@@ -20,26 +95,10 @@ export const TOOL_DEFINITIONS = [
20
95
  properties: {
21
96
  model: {
22
97
  type: 'string',
23
- description: 'Model ID (optional, uses default). Append `:nitro` for the fastest variant, ' +
24
- '`:floor` for the cheapest, `:free` for the free tier, `:online` for web search, ' +
25
- 'or `:exacto` for the best tool-calling accuracy. ' +
26
- 'Example: `openai/gpt-4o:nitro`. Or pass `online: true` for programmatic web search control.',
98
+ description: 'Model ID (optional, uses server default). Append `:nitro` (fastest), `:floor` (cheapest), `:free`, `:online` (web search), or `:exacto` (tool accuracy). Example: `openai/gpt-4o:nitro`. Or pass `online: true` for programmatic web search.',
27
99
  },
28
- messages: {
29
- type: 'array',
30
- minItems: 1,
31
- items: {
32
- type: 'object',
33
- properties: {
34
- role: { type: 'string', enum: ['system', 'user', 'assistant'] },
35
- content: {
36
- oneOf: [{ type: 'string' }, { type: 'array', items: { type: 'object' } }],
37
- },
38
- },
39
- required: ['role', 'content'],
40
- },
41
- },
42
- temperature: { type: 'number', minimum: 0, maximum: 2 },
100
+ messages: CHAT_MESSAGE_SCHEMA,
101
+ temperature: { type: 'number', minimum: 0, maximum: 2, description: 'Default: 1.' },
43
102
  max_tokens: {
44
103
  type: 'number',
45
104
  minimum: 1,
@@ -47,8 +106,7 @@ export const TOOL_DEFINITIONS = [
47
106
  },
48
107
  provider: {
49
108
  type: 'object',
50
- description: 'OpenRouter provider-routing overrides. Merges on top of `OPENROUTER_PROVIDER_*` env defaults. ' +
51
- 'See https://openrouter.ai/docs/features/provider-routing',
109
+ description: 'OpenRouter provider-routing overrides. Merges on top of `OPENROUTER_PROVIDER_*` env defaults. See https://openrouter.ai/docs/features/provider-routing',
52
110
  properties: {
53
111
  quantizations: {
54
112
  type: 'array',
@@ -60,19 +118,16 @@ export const TOOL_DEFINITIONS = [
60
118
  items: { type: 'string' },
61
119
  description: 'Exclude these provider slugs.',
62
120
  },
63
- sort: {
64
- type: 'string',
65
- enum: ['price', 'throughput', 'latency'],
66
- },
121
+ sort: { type: 'string', enum: [...PROVIDER_SORT_VALUES] },
67
122
  order: { type: 'array', items: { type: 'string' } },
68
123
  require_parameters: { type: 'boolean' },
69
- data_collection: { type: 'string', enum: ['allow', 'deny'] },
124
+ data_collection: { type: 'string', enum: [...PROVIDER_DATA_COLLECTION] },
70
125
  allow_fallbacks: { type: 'boolean' },
71
126
  },
72
127
  },
73
128
  include_reasoning: {
74
129
  type: 'boolean',
75
- description: "Surface the model's chain-of-thought on `_meta.reasoning` for R1 / Opus 4.7 / Gemini Thinking.",
130
+ description: "Surface the model's chain-of-thought on `_meta.reasoning` for R1 / Opus / Gemini Thinking models.",
76
131
  },
77
132
  online: {
78
133
  type: 'boolean',
@@ -83,19 +138,7 @@ export const TOOL_DEFINITIONS = [
83
138
  minimum: 1,
84
139
  description: 'Max web-search results when `online: true` (default 5).',
85
140
  },
86
- cache: {
87
- type: 'boolean',
88
- description: 'Enable OpenRouter response caching via `X-OpenRouter-Cache: true`. ' +
89
- 'Server-wide default settable via `OPENROUTER_CACHE_RESPONSES=1`.',
90
- },
91
- cache_ttl: {
92
- type: 'string',
93
- description: 'Cache TTL (e.g. `"5m"`, `"1h"`, `"24h"`; 1s-24h range).',
94
- },
95
- cache_clear: {
96
- type: 'boolean',
97
- description: 'Bust the cache entry for this exact request.',
98
- },
141
+ ...CACHE_PROPERTIES,
99
142
  },
100
143
  required: ['messages'],
101
144
  },
@@ -113,30 +156,15 @@ export const TOOL_DEFINITIONS = [
113
156
  inputSchema: {
114
157
  type: 'object',
115
158
  properties: {
116
- model: { type: 'string', description: 'Model ID (same as chat_completion).' },
117
- messages: {
118
- type: 'array',
119
- minItems: 1,
120
- items: {
121
- type: 'object',
122
- properties: {
123
- role: { type: 'string', enum: ['system', 'user', 'assistant'] },
124
- content: {
125
- oneOf: [{ type: 'string' }, { type: 'array', items: { type: 'object' } }],
126
- },
127
- },
128
- required: ['role', 'content'],
129
- },
130
- },
159
+ model: { type: 'string', description: 'Model ID (same options as chat_completion).' },
160
+ messages: CHAT_MESSAGE_SCHEMA,
131
161
  temperature: { type: 'number', minimum: 0, maximum: 2 },
132
162
  max_tokens: { type: 'number', minimum: 1 },
133
163
  provider: { type: 'object' },
134
164
  include_reasoning: { type: 'boolean' },
135
165
  online: { type: 'boolean' },
136
166
  web_max_results: { type: 'number', minimum: 1 },
137
- cache: { type: 'boolean' },
138
- cache_ttl: { type: 'string' },
139
- cache_clear: { type: 'boolean' },
167
+ ...CACHE_PROPERTIES,
140
168
  },
141
169
  required: ['messages'],
142
170
  },
@@ -177,23 +205,21 @@ export const TOOL_DEFINITIONS = [
177
205
  properties: {
178
206
  image_path: {
179
207
  type: 'string',
180
- description: 'Required. Local path (inside OPENROUTER_INPUT_DIR sandbox), https URL, or data URL. ' +
181
- 'Good: `"photo.jpg"`. Bad: `"url": "..."` (wrong key), `"/etc/passwd"` (UNSAFE_PATH).',
208
+ description: 'Local path (OPENROUTER_INPUT_DIR sandbox), https URL, or data URL. Good: `"photo.jpg"`. Bad: `"url": "..."` (wrong key), `"/etc/passwd"` (UNSAFE_PATH).',
182
209
  },
183
210
  question: {
184
211
  type: 'string',
185
- description: 'Optional question about the image. Defaults to "What\'s in this image?" if omitted. ' +
186
- 'Good: `"List all text"`. Bad: using `prompt` key (wrong name for this tool).',
212
+ description: 'Optional question. Default: "What\'s in this image?". Bad: using `prompt` (wrong key for this tool).',
213
+ },
214
+ model: {
215
+ type: 'string',
216
+ description: 'Vision model ID (optional; server default is a free multimodal model).',
187
217
  },
188
- model: { type: 'string' },
189
218
  cache_input: {
190
219
  type: 'boolean',
191
- description: 'Attach `cache_control: ephemeral` to the image block so Anthropic / Gemini prompt-cache it. ' +
192
- 'Repeat questions about the same image save ~10x on Anthropic.',
220
+ description: 'Attach `cache_control: ephemeral` to the image block for Anthropic / Gemini prompt caching.',
193
221
  },
194
- cache: { type: 'boolean' },
195
- cache_ttl: { type: 'string' },
196
- cache_clear: { type: 'boolean' },
222
+ ...CACHE_PROPERTIES,
197
223
  },
198
224
  required: ['image_path'],
199
225
  },
@@ -213,18 +239,18 @@ export const TOOL_DEFINITIONS = [
213
239
  properties: {
214
240
  audio_path: {
215
241
  type: 'string',
216
- description: 'Local file path (sandboxed to OPENROUTER_INPUT_DIR / OPENROUTER_OUTPUT_DIR / cwd), ' +
217
- 'http(s) URL, or data URL (base64-encoded audio)',
242
+ description: 'Local file path (sandboxed), http(s) URL, or data URL (base64-encoded audio).',
218
243
  },
219
244
  question: {
220
245
  type: 'string',
221
- description: 'Question or instruction about the audio (default: transcribe)',
246
+ description: 'Question or instruction. Default: "Please transcribe and analyze this audio file."',
247
+ },
248
+ model: {
249
+ type: 'string',
250
+ description: 'Multimodal model ID (default: google/gemini-2.5-flash).',
222
251
  },
223
- model: { type: 'string' },
224
252
  cache_input: { type: 'boolean' },
225
- cache: { type: 'boolean' },
226
- cache_ttl: { type: 'string' },
227
- cache_clear: { type: 'boolean' },
253
+ ...CACHE_PROPERTIES,
228
254
  },
229
255
  required: ['audio_path'],
230
256
  },
@@ -244,15 +270,18 @@ export const TOOL_DEFINITIONS = [
244
270
  properties: {
245
271
  video_path: {
246
272
  type: 'string',
247
- description: 'Local file path (sandboxed to OPENROUTER_INPUT_DIR / OPENROUTER_OUTPUT_DIR / cwd), ' +
248
- 'http(s) URL, or base64 data URL. Supported: mp4 / mpeg / mov / webm.',
273
+ description: 'Local file path (sandboxed), http(s) URL, or base64 data URL. Supported containers: mp4, mpeg, mov, webm.',
274
+ },
275
+ question: {
276
+ type: 'string',
277
+ description: 'Optional question. Default: "Describe what happens in this video, step by step."',
278
+ },
279
+ model: {
280
+ type: 'string',
281
+ description: 'Video-capable model ID (default: google/gemini-2.5-flash).',
249
282
  },
250
- question: { type: 'string' },
251
- model: { type: 'string' },
252
283
  cache_input: { type: 'boolean' },
253
- cache: { type: 'boolean' },
254
- cache_ttl: { type: 'string' },
255
- cache_clear: { type: 'boolean' },
284
+ ...CACHE_PROPERTIES,
256
285
  },
257
286
  required: ['video_path'],
258
287
  },
@@ -270,8 +299,11 @@ export const TOOL_DEFINITIONS = [
270
299
  inputSchema: {
271
300
  type: 'object',
272
301
  properties: {
273
- query: { type: 'string' },
274
- provider: { type: 'string' },
302
+ query: { type: 'string', description: 'Substring match against model id or name.' },
303
+ provider: {
304
+ type: 'string',
305
+ description: 'Filter by provider slug prefix (e.g. `google`).',
306
+ },
275
307
  capabilities: {
276
308
  type: 'object',
277
309
  properties: {
@@ -280,8 +312,13 @@ export const TOOL_DEFINITIONS = [
280
312
  video: { type: 'boolean' },
281
313
  },
282
314
  },
283
- limit: { type: 'number', minimum: 1, maximum: 50 },
284
- offset: { type: 'number', minimum: 0 },
315
+ limit: {
316
+ type: 'number',
317
+ minimum: 1,
318
+ maximum: 50,
319
+ description: 'Page size (default 20, max 50).',
320
+ },
321
+ offset: { type: 'number', minimum: 0, description: 'Pagination offset (default 0).' },
285
322
  },
286
323
  },
287
324
  outputSchema: {
@@ -309,7 +346,12 @@ export const TOOL_DEFINITIONS = [
309
346
  },
310
347
  inputSchema: {
311
348
  type: 'object',
312
- properties: { model: { type: 'string' } },
349
+ properties: {
350
+ model: {
351
+ type: 'string',
352
+ description: 'Full OpenRouter model slug (e.g. `openai/gpt-4o`).',
353
+ },
354
+ },
313
355
  required: ['model'],
314
356
  },
315
357
  outputSchema: {
@@ -335,7 +377,9 @@ export const TOOL_DEFINITIONS = [
335
377
  },
336
378
  inputSchema: {
337
379
  type: 'object',
338
- properties: { model: { type: 'string' } },
380
+ properties: {
381
+ model: { type: 'string', description: 'Full OpenRouter model slug to check.' },
382
+ },
339
383
  required: ['model'],
340
384
  },
341
385
  outputSchema: {
@@ -360,32 +404,33 @@ export const TOOL_DEFINITIONS = [
360
404
  inputSchema: {
361
405
  type: 'object',
362
406
  properties: {
363
- prompt: { type: 'string' },
364
- model: { type: 'string' },
407
+ prompt: { type: 'string', description: 'Text prompt describing the image to generate.' },
408
+ model: {
409
+ type: 'string',
410
+ description: 'Image model ID (default: google/gemini-2.5-flash-image). Chat-completions route.',
411
+ },
365
412
  aspect_ratio: {
366
413
  type: 'string',
367
- enum: [
368
- '1:1',
369
- '2:3',
370
- '3:2',
371
- '3:4',
372
- '4:3',
373
- '4:5',
374
- '5:4',
375
- '9:16',
376
- '16:9',
377
- '21:9',
378
- '1:4',
379
- '4:1',
380
- '1:8',
381
- '8:1',
382
- ],
383
- },
384
- image_size: { type: 'string', enum: ['0.5K', '1K', '2K', '4K'] },
385
- max_tokens: { type: 'number', minimum: 1 },
414
+ enum: [...IMAGE_ASPECT_RATIOS],
415
+ description: 'Optional aspect ratio (provider-dependent).',
416
+ },
417
+ image_size: {
418
+ type: 'string',
419
+ enum: [...IMAGE_SIZES],
420
+ description: 'Optional resolution tier for supported models.',
421
+ },
422
+ max_tokens: { type: 'number', minimum: 1, description: 'Optional completion token cap.' },
386
423
  save_path: SAVE_PATH_PROPERTY,
387
- input_images: { type: 'array', items: { type: 'string' } },
388
- modalities: { type: 'array', items: { type: 'string' } },
424
+ input_images: {
425
+ type: 'array',
426
+ items: { type: 'string' },
427
+ description: 'Reference images (local path, URL, or data URL) for style/identity conditioning.',
428
+ },
429
+ modalities: {
430
+ type: 'array',
431
+ items: { type: 'string' },
432
+ description: 'Response modalities (default: `["image","text"]`).',
433
+ },
389
434
  },
390
435
  required: ['prompt'],
391
436
  },
@@ -409,49 +454,45 @@ export const TOOL_DEFINITIONS = [
409
454
  },
410
455
  model: {
411
456
  type: 'string',
412
- description: 'Image model ID. Default: google/gemini-2.5-flash-image. Browse available at https://openrouter.ai/collections/image-models',
457
+ description: 'Image model ID. Default: google/gemini-2.5-flash-image. Browse: https://openrouter.ai/collections/image-models',
413
458
  },
414
459
  resolution: {
415
460
  type: 'string',
416
- enum: ['512', '0.5K', '1K', '2K', '4K'],
461
+ enum: [...IMAGE_DEDICATED_RESOLUTIONS],
417
462
  description: 'Normalized resolution tier. Provider maps to closest supported size.',
418
463
  },
419
464
  aspect_ratio: {
420
465
  type: 'string',
421
- description: 'Aspect ratio (e.g. "1:1", "16:9", "9:16", "4:3", "21:9").',
466
+ enum: [...IMAGE_ASPECT_RATIOS],
467
+ description: 'Aspect ratio (same enum as generate_image).',
422
468
  },
423
469
  quality: {
424
470
  type: 'string',
425
- enum: ['auto', 'low', 'medium', 'high'],
426
- description: 'Image quality level.',
471
+ enum: [...IMAGE_DEDICATED_QUALITIES],
472
+ description: 'Image quality level (default: auto).',
427
473
  },
428
474
  output_format: {
429
475
  type: 'string',
430
- enum: ['png', 'jpeg', 'webp', 'svg'],
476
+ enum: [...IMAGE_OUTPUT_FORMATS],
431
477
  description: 'Output image format.',
432
478
  },
433
479
  n: {
434
480
  type: 'number',
435
481
  minimum: 1,
436
482
  maximum: 10,
437
- description: 'Number of images to generate (model-dependent, default 1).',
483
+ description: 'Number of images to request (default 1; only images[0] is saved/inlined).',
438
484
  },
439
485
  input_references: {
440
486
  type: 'array',
441
487
  items: { type: 'string' },
442
- description: 'Reference images for image-to-image workflows. Each entry: local path, http(s) URL, or data URL.',
443
- },
444
- save_path: {
445
- ...SAVE_PATH_PROPERTY,
446
- description: 'Save generated image to this path. ' + SAVE_PATH_PROPERTY.description,
488
+ description: 'Reference images for image-to-image. Each entry: local path, http(s) URL, or data URL.',
447
489
  },
490
+ save_path: SAVE_PATH_WITH_PREFIX('Save generated image to this path.'),
448
491
  provider: {
449
492
  type: 'object',
450
493
  description: 'Provider routing overrides (order, sort, allow_fallbacks, etc.).',
451
494
  },
452
- cache: { type: 'boolean' },
453
- cache_ttl: { type: 'string' },
454
- cache_clear: { type: 'boolean' },
495
+ ...CACHE_PROPERTIES,
455
496
  },
456
497
  required: ['prompt'],
457
498
  },
@@ -469,10 +510,20 @@ export const TOOL_DEFINITIONS = [
469
510
  inputSchema: {
470
511
  type: 'object',
471
512
  properties: {
472
- prompt: { type: 'string' },
473
- model: { type: 'string' },
474
- voice: { type: 'string' },
475
- format: { type: 'string' },
513
+ prompt: { type: 'string', description: 'Text prompt for speech or music generation.' },
514
+ model: {
515
+ type: 'string',
516
+ description: 'Chat-completions audio model (default: openai/gpt-audio).',
517
+ },
518
+ voice: {
519
+ type: 'string',
520
+ description: 'Voice ID (default: alloy). Model-specific.',
521
+ },
522
+ format: {
523
+ type: 'string',
524
+ enum: [...GENERATE_AUDIO_FORMATS],
525
+ description: 'Output audio format (default: pcm16, auto-wrapped as WAV when needed).',
526
+ },
476
527
  save_path: SAVE_PATH_PROPERTY,
477
528
  },
478
529
  required: ['prompt'],
@@ -505,26 +556,21 @@ export const TOOL_DEFINITIONS = [
505
556
  },
506
557
  response_format: {
507
558
  type: 'string',
508
- enum: ['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm'],
559
+ enum: [...TTS_RESPONSE_FORMATS],
509
560
  description: 'Output audio format. Default: mp3.',
510
561
  },
511
562
  speed: {
512
563
  type: 'number',
513
564
  minimum: 0.25,
514
565
  maximum: 4.0,
515
- description: 'Speed of speech (0.25 to 4.0). Default: 1.0.',
566
+ description: 'Speed of speech (0.25–4.0). Default: 1.0.',
516
567
  },
517
568
  instructions: {
518
569
  type: 'string',
519
570
  description: 'Tone/style instructions (e.g. "speak in a warm, friendly tone"). OpenAI models only.',
520
571
  },
521
- save_path: {
522
- ...SAVE_PATH_PROPERTY,
523
- description: 'Save audio to this path. ' + SAVE_PATH_PROPERTY.description,
524
- },
525
- cache: { type: 'boolean' },
526
- cache_ttl: { type: 'string' },
527
- cache_clear: { type: 'boolean' },
572
+ save_path: SAVE_PATH_WITH_PREFIX('Save audio to this path.'),
573
+ ...CACHE_PROPERTIES,
528
574
  },
529
575
  required: ['input'],
530
576
  },
@@ -544,7 +590,7 @@ export const TOOL_DEFINITIONS = [
544
590
  properties: {
545
591
  audio_path: {
546
592
  type: 'string',
547
- description: 'Audio file: local path (sandboxed), http(s) URL, or base64 data URL. Formats: mp3, wav, flac, ogg, webm, mp4.',
593
+ description: 'Audio file: local path (sandboxed), http(s) URL, or base64 data URL. Formats: mp3, wav, flac, ogg, webm, mp4, m4a.',
548
594
  },
549
595
  model: {
550
596
  type: 'string',
@@ -556,18 +602,16 @@ export const TOOL_DEFINITIONS = [
556
602
  },
557
603
  response_format: {
558
604
  type: 'string',
559
- enum: ['json', 'text', 'srt', 'verbose_json', 'vtt'],
605
+ enum: [...STT_RESPONSE_FORMATS],
560
606
  description: 'Output format for transcription. Default: json.',
561
607
  },
562
608
  temperature: {
563
609
  type: 'number',
564
610
  minimum: 0,
565
611
  maximum: 1,
566
- description: 'Sampling temperature for transcription (0-1).',
612
+ description: 'Sampling temperature for transcription (0–1).',
567
613
  },
568
- cache: { type: 'boolean' },
569
- cache_ttl: { type: 'string' },
570
- cache_clear: { type: 'boolean' },
614
+ ...CACHE_PROPERTIES,
571
615
  },
572
616
  required: ['audio_path'],
573
617
  },
@@ -585,19 +629,50 @@ export const TOOL_DEFINITIONS = [
585
629
  inputSchema: {
586
630
  type: 'object',
587
631
  properties: {
588
- prompt: { type: 'string' },
589
- model: { type: 'string' },
590
- resolution: { type: 'string' },
591
- aspect_ratio: { type: 'string' },
592
- duration: { type: 'number', minimum: 1 },
593
- seed: { type: 'number' },
594
- first_frame_image: { type: 'string' },
595
- last_frame_image: { type: 'string' },
596
- reference_images: { type: 'array', items: { type: 'string' } },
597
- provider: { type: 'object' },
632
+ prompt: { type: 'string', description: 'Text prompt describing the video to generate.' },
633
+ model: {
634
+ type: 'string',
635
+ description: 'Video model ID (default: google/veo-3.1). Passed through to provider.',
636
+ },
637
+ resolution: {
638
+ type: 'string',
639
+ description: 'Provider-specific resolution (e.g. "720p", "1080p"). No server-side enum.',
640
+ },
641
+ aspect_ratio: {
642
+ type: 'string',
643
+ description: 'Provider-specific aspect ratio (e.g. "16:9", "9:16"). No server-side enum.',
644
+ },
645
+ duration: {
646
+ type: 'number',
647
+ minimum: 1,
648
+ description: 'Clip duration in seconds (provider-dependent).',
649
+ },
650
+ seed: { type: 'number', description: 'Optional reproducibility seed.' },
651
+ first_frame_image: {
652
+ type: 'string',
653
+ description: 'Optional first-frame image (path, URL, or data URL).',
654
+ },
655
+ last_frame_image: {
656
+ type: 'string',
657
+ description: 'Optional last-frame image (path, URL, or data URL).',
658
+ },
659
+ reference_images: {
660
+ type: 'array',
661
+ items: { type: 'string' },
662
+ description: 'Optional reference images for style/subject guidance.',
663
+ },
664
+ provider: { type: 'object', description: 'Provider routing overrides.' },
598
665
  save_path: SAVE_PATH_PROPERTY,
599
- max_wait_ms: { type: 'number', minimum: 10000 },
600
- poll_interval_ms: { type: 'number', minimum: 2000 },
666
+ max_wait_ms: {
667
+ type: 'number',
668
+ minimum: 100,
669
+ description: 'Max time to poll before returning JOB_STILL_RUNNING (ms). Default: 600000 (10 min) via OPENROUTER_VIDEO_MAX_WAIT_MS.',
670
+ },
671
+ poll_interval_ms: {
672
+ type: 'number',
673
+ minimum: 50,
674
+ description: 'Poll interval while waiting (ms). Default: 15000 via OPENROUTER_VIDEO_POLL_INTERVAL_MS.',
675
+ },
601
676
  },
602
677
  required: ['prompt'],
603
678
  },
@@ -619,15 +694,23 @@ export const TOOL_DEFINITIONS = [
619
694
  type: 'string',
620
695
  description: 'First-frame image (path, URL, or data URL). Required.',
621
696
  },
622
- prompt: { type: 'string' },
623
- model: { type: 'string' },
624
- resolution: { type: 'string' },
625
- aspect_ratio: { type: 'string' },
626
- duration: { type: 'number', minimum: 1 },
627
- seed: { type: 'number' },
697
+ prompt: { type: 'string', description: 'Motion/scene prompt describing the video.' },
698
+ model: { type: 'string', description: 'Video model ID (default: google/veo-3.1).' },
699
+ resolution: { type: 'string', description: 'Provider-specific resolution.' },
700
+ aspect_ratio: { type: 'string', description: 'Provider-specific aspect ratio.' },
701
+ duration: { type: 'number', minimum: 1, description: 'Clip duration in seconds.' },
702
+ seed: { type: 'number', description: 'Optional reproducibility seed.' },
628
703
  save_path: SAVE_PATH_PROPERTY,
629
- max_wait_ms: { type: 'number', minimum: 10000 },
630
- poll_interval_ms: { type: 'number', minimum: 2000 },
704
+ max_wait_ms: {
705
+ type: 'number',
706
+ minimum: 100,
707
+ description: 'Max poll wait (ms) before JOB_STILL_RUNNING. Default: 600000.',
708
+ },
709
+ poll_interval_ms: {
710
+ type: 'number',
711
+ minimum: 50,
712
+ description: 'Poll interval (ms). Default: 15000.',
713
+ },
631
714
  },
632
715
  required: ['image', 'prompt'],
633
716
  },
@@ -645,7 +728,10 @@ export const TOOL_DEFINITIONS = [
645
728
  inputSchema: {
646
729
  type: 'object',
647
730
  properties: {
648
- video_id: { type: 'string' },
731
+ video_id: {
732
+ type: 'string',
733
+ description: 'Video job id from generate_video / generate_video_from_image.',
734
+ },
649
735
  save_path: SAVE_PATH_PROPERTY,
650
736
  },
651
737
  required: ['video_id'],
@@ -664,11 +750,17 @@ export const TOOL_DEFINITIONS = [
664
750
  inputSchema: {
665
751
  type: 'object',
666
752
  properties: {
667
- query: { type: 'string' },
753
+ query: { type: 'string', description: 'Search query to rank documents against.' },
668
754
  documents: { type: 'array', items: { type: 'string' }, minItems: 1 },
669
- model: { type: 'string' },
670
- top_n: { type: 'number', minimum: 1 },
671
- return_documents: { type: 'boolean' },
755
+ model: {
756
+ type: 'string',
757
+ description: 'Reranker model (default: cohere/rerank-english-v3.0).',
758
+ },
759
+ top_n: { type: 'number', minimum: 1, description: 'Return only the top N results.' },
760
+ return_documents: {
761
+ type: 'boolean',
762
+ description: 'When true, include original document text in each result.',
763
+ },
672
764
  },
673
765
  required: ['query', 'documents'],
674
766
  },
@@ -717,3 +809,13 @@ export const TOOL_DEFINITIONS = [
717
809
  },
718
810
  },
719
811
  ];
812
+ /** Tool names from definitions — must match TOOL_NAMES in tool-descriptions.ts. */
813
+ export const TOOL_DEFINITION_NAMES = TOOL_DEFINITIONS.map((t) => t.name);
814
+ if (TOOL_DEFINITION_NAMES.length !== TOOL_NAMES.length) {
815
+ throw new Error(`Tool count mismatch: definitions=${TOOL_DEFINITION_NAMES.length} descriptions=${TOOL_NAMES.length}`);
816
+ }
817
+ for (const name of TOOL_NAMES) {
818
+ if (!TOOL_DEFINITION_NAMES.includes(name)) {
819
+ throw new Error(`Missing tool definition for ${name}`);
820
+ }
821
+ }