@stabgan/openrouter-mcp-multimodal 4.5.0 → 4.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/README.md +380 -242
  2. package/dist/index.js +22 -6
  3. package/dist/model-cache.d.ts +35 -12
  4. package/dist/model-cache.js +79 -22
  5. package/dist/tool-descriptions.d.ts +19 -0
  6. package/dist/tool-descriptions.js +423 -0
  7. package/dist/tool-handlers/analyze-audio.js +5 -1
  8. package/dist/tool-handlers/analyze-image.js +6 -5
  9. package/dist/tool-handlers/analyze-video.js +6 -5
  10. package/dist/tool-handlers/audio-utils.js +4 -2
  11. package/dist/tool-handlers/chat-completion.js +1 -1
  12. package/dist/tool-handlers/fetch-utils.js +16 -2
  13. package/dist/tool-handlers/generate-audio.js +2 -4
  14. package/dist/tool-handlers/generate-image-input.d.ts +3 -0
  15. package/dist/tool-handlers/generate-image-input.js +38 -0
  16. package/dist/tool-handlers/generate-image.d.ts +13 -51
  17. package/dist/tool-handlers/generate-image.js +32 -119
  18. package/dist/tool-handlers/generate-video.js +28 -24
  19. package/dist/tool-handlers/health-check.js +4 -1
  20. package/dist/tool-handlers/image-utils.d.ts +1 -0
  21. package/dist/tool-handlers/image-utils.js +26 -16
  22. package/dist/tool-handlers/openrouter-errors.d.ts +5 -1
  23. package/dist/tool-handlers/openrouter-errors.js +78 -13
  24. package/dist/tool-handlers/provider-routing.js +7 -2
  25. package/dist/tool-handlers/rerank.js +2 -5
  26. package/dist/tool-handlers/search-models.d.ts +2 -2
  27. package/dist/tool-handlers/search-models.js +2 -6
  28. package/dist/tool-handlers/structured-output.d.ts +8 -0
  29. package/dist/tool-handlers/structured-output.js +11 -0
  30. package/dist/tool-handlers/video-utils.js +6 -9
  31. package/dist/tool-handlers.js +43 -123
  32. package/dist/version.d.ts +1 -1
  33. package/dist/version.js +1 -1
  34. package/package.json +26 -14
@@ -0,0 +1,423 @@
1
+ /**
2
+ * MCP tool descriptions with explicit routing, examples, and failure modes.
3
+ * See docs/plans/tool-description-improvement.md for the authoring guide.
4
+ */
5
+ function formatBullets(items) {
6
+ return items.map((item) => `- ${item}`).join('\n');
7
+ }
8
+ export function buildToolDescription(parts) {
9
+ return (`${parts.summary}\n\n` +
10
+ `Use when:\n${formatBullets(parts.useWhen)}\n\n` +
11
+ `Do NOT use when:\n${formatBullets(parts.notWhen)}\n\n` +
12
+ `Good examples:\n${formatBullets(parts.goodExamples)}\n\n` +
13
+ `Bad examples:\n${formatBullets(parts.badExamples)}\n\n` +
14
+ `Fails when:\n${formatBullets(parts.failsWhen)}\n\n` +
15
+ `Works with: ${parts.worksWith.join(', ')}.`);
16
+ }
17
+ /** Required sections every tool description must contain (regression-tested). */
18
+ export const REQUIRED_DESCRIPTION_SECTIONS = [
19
+ 'Use when:',
20
+ 'Do NOT use when:',
21
+ 'Good examples:',
22
+ 'Bad examples:',
23
+ 'Fails when:',
24
+ 'Works with:',
25
+ ];
26
+ export const TOOL_NAMES = [
27
+ 'chat_completion',
28
+ 'analyze_image',
29
+ 'analyze_audio',
30
+ 'analyze_video',
31
+ 'search_models',
32
+ 'get_model_info',
33
+ 'validate_model',
34
+ 'generate_image',
35
+ 'generate_audio',
36
+ 'generate_video',
37
+ 'generate_video_from_image',
38
+ 'get_video_status',
39
+ 'rerank_documents',
40
+ 'health_check',
41
+ ];
42
+ export const TOOL_DESCRIPTIONS = {
43
+ chat_completion: buildToolDescription({
44
+ summary: 'Send messages to an OpenRouter chat model and get a text reply. Supports provider routing, ' +
45
+ 'model suffixes (`:nitro` fastest, `:floor` cheapest, `:exacto` tool accuracy), reasoning ' +
46
+ 'tokens, web search (`online: true`), and response caching.',
47
+ useWhen: [
48
+ 'You need text generation, Q&A, summarization, or multi-turn dialogue',
49
+ 'You want web-grounded answers (`online: true`)',
50
+ 'You already know the model id (or rely on the server default)',
51
+ ],
52
+ notWhen: [
53
+ 'Input is an image/audio/video file → use analyze_image / analyze_audio / analyze_video',
54
+ 'You need to create images, audio, or video → use generate_* tools',
55
+ 'You only need to check if a model exists → use validate_model',
56
+ ],
57
+ goodExamples: [
58
+ '`{ "messages": [{ "role": "user", "content": "Explain recursion in one paragraph." }] }`',
59
+ '`{ "model": "openai/gpt-4o:nitro", "messages": [...], "online": true }` for web search',
60
+ '`{ "messages": [...], "include_reasoning": true }` for chain-of-thought models',
61
+ ],
62
+ badExamples: [
63
+ '`{ "messages": [] }` → INVALID_INPUT (empty array)',
64
+ '`{ "image_path": "photo.jpg" }` → wrong tool; use analyze_image',
65
+ 'Putting file paths inside message content without a vision model configured',
66
+ ],
67
+ failsWhen: [
68
+ 'INVALID_INPUT: messages array is empty',
69
+ 'UPSTREAM_REFUSED: credits, content policy, or rate limit',
70
+ 'UPSTREAM_TIMEOUT: upstream did not respond in time',
71
+ 'MODEL_NOT_FOUND: model slug does not exist on OpenRouter',
72
+ ],
73
+ worksWith: ['validate_model', 'search_models'],
74
+ }),
75
+ analyze_image: buildToolDescription({
76
+ summary: 'Analyze one image with a vision model. Accepts a sandboxed local path, https URL, or base64 data URL. ' +
77
+ 'Output is model-generated and tagged `_meta.content_is_untrusted: true`.',
78
+ useWhen: [
79
+ 'You have one image and need OCR, captioning, or visual Q&A',
80
+ 'The image is a local file under the input sandbox, a public https URL, or a data URL',
81
+ ],
82
+ notWhen: [
83
+ 'You want to generate a new image → use generate_image',
84
+ 'You need multi-file batch analysis in one call → not supported; call once per image',
85
+ 'Pure text chat → use chat_completion with a vision-capable model instead (less ergonomic)',
86
+ ],
87
+ goodExamples: [
88
+ '`{ "image_path": "diagram.png", "question": "List every label in this diagram." }`',
89
+ '`{ "image_path": "https://example.com/photo.jpg", "question": "Describe the scene." }`',
90
+ '`{ "model": "google/gemini-2.5-flash", "image_path": "scan.jpg", "question": "Extract text" }`',
91
+ ],
92
+ badExamples: [
93
+ '`{ "url": "photo.jpg" }` → wrong key; use `image_path`',
94
+ '`{ "prompt": "describe" }` → wrong key; use `question` (optional, defaults to "What\'s in this image?")',
95
+ '`{ "image_path": "../../../etc/passwd" }` → UNSAFE_PATH (sandbox escape)',
96
+ ],
97
+ failsWhen: [
98
+ 'INVALID_INPUT: image_path missing or malformed',
99
+ 'UNSAFE_PATH: local path escaped the input sandbox',
100
+ 'RESOURCE_TOO_LARGE: image exceeded fetch size cap',
101
+ 'UPSTREAM_REFUSED: SSRF block, bad URL, or content policy',
102
+ ],
103
+ worksWith: ['search_models', 'generate_image'],
104
+ }),
105
+ analyze_audio: buildToolDescription({
106
+ summary: 'Transcribe or analyze one audio file (WAV, MP3, FLAC, OGG, etc.) with a multimodal model. ' +
107
+ 'Output is tagged `_meta.content_is_untrusted: true`.',
108
+ useWhen: [
109
+ 'You have a local audio file or URL and need transcription or audio understanding',
110
+ 'Format is a common audio container the decoder recognizes',
111
+ ],
112
+ notWhen: [
113
+ 'You want text-to-speech → use generate_audio',
114
+ 'Input is video → use analyze_video (or extract audio first)',
115
+ 'Pure text chat → use chat_completion',
116
+ ],
117
+ goodExamples: [
118
+ '`{ "audio_path": "meeting.wav", "question": "Transcribe verbatim." }`',
119
+ '`{ "audio_path": "https://example.com/podcast.mp3", "question": "Summarize topics." }`',
120
+ ],
121
+ badExamples: [
122
+ '`{ "audio_path": "/etc/shadow" }` → UNSAFE_PATH',
123
+ '`{ "path": "song.mp3" }` → wrong key; use `audio_path`',
124
+ 'Non-audio binary renamed to .mp3 → UNSUPPORTED_FORMAT',
125
+ ],
126
+ failsWhen: [
127
+ 'INVALID_INPUT: audio_path missing',
128
+ 'UNSAFE_PATH: local path escaped the sandbox',
129
+ 'UNSUPPORTED_FORMAT: file is not recognized as audio',
130
+ 'RESOURCE_TOO_LARGE: exceeds size cap',
131
+ ],
132
+ worksWith: ['generate_audio', 'search_models'],
133
+ }),
134
+ analyze_video: buildToolDescription({
135
+ summary: 'Describe or analyze one video file (mp4, mpeg, mov, webm). Default model: google/gemini-2.5-flash. ' +
136
+ 'Output is tagged `_meta.content_is_untrusted: true`. Large files are fully buffered — prefer short clips.',
137
+ useWhen: [
138
+ 'You need a summary, scene description, or Q&A over a video file',
139
+ 'Video is within size limits and readable by the decoder',
140
+ ],
141
+ notWhen: [
142
+ 'You want to generate video → use generate_video',
143
+ 'You only need audio → use analyze_audio',
144
+ 'Video is very large → trim first or expect RESOURCE_TOO_LARGE',
145
+ ],
146
+ goodExamples: [
147
+ '`{ "video_path": "clip.mp4", "question": "What happens in the first 30 seconds?" }`',
148
+ '`{ "video_path": "demo.webm", "question": "List on-screen text." }`',
149
+ ],
150
+ badExamples: [
151
+ '`{ "video_path": "../secret.mp4" }` → UNSAFE_PATH',
152
+ 'Expecting frame-by-frame timestamps without asking in the prompt',
153
+ 'Using analyze_video for async generation status → use get_video_status',
154
+ ],
155
+ failsWhen: [
156
+ 'INVALID_INPUT: video_path missing',
157
+ 'UNSAFE_PATH: path escaped the sandbox',
158
+ 'UNSUPPORTED_FORMAT: unrecognized video container',
159
+ 'RESOURCE_TOO_LARGE: exceeds fetch cap',
160
+ ],
161
+ worksWith: ['generate_video', 'get_video_status', 'search_models'],
162
+ }),
163
+ search_models: buildToolDescription({
164
+ summary: 'Search the OpenRouter model catalog by name, provider, or capability. Returns a paginated slice; ' +
165
+ 'use `offset`, `limit`, and `next_offset` to page through large result sets.',
166
+ useWhen: [
167
+ 'You do not know which model id to use',
168
+ 'You need vision/audio/video-capable models filtered by modality',
169
+ 'You want models from a specific provider prefix (e.g. `google`)',
170
+ ],
171
+ notWhen: [
172
+ 'You already have a model id and only need existence check → validate_model',
173
+ 'You need pricing/context details for one id → get_model_info',
174
+ 'You expect all 400+ models in one response without paging',
175
+ ],
176
+ goodExamples: [
177
+ '`{ "query": "gemini", "capabilities": { "vision": true }, "limit": 10, "offset": 0 }`',
178
+ '`{ "provider": "anthropic", "limit": 20 }`',
179
+ 'Page 2: `{ "query": "llama", "offset": 20, "limit": 20 }` using prior `next_offset`',
180
+ ],
181
+ badExamples: [
182
+ 'Omitting pagination on broad queries → large payload; use limit/offset',
183
+ 'Using search_models output as chat messages → use returned `id` in chat_completion',
184
+ '`{ "capability": "vision" }` → wrong shape; use `capabilities: { "vision": true }`',
185
+ ],
186
+ failsWhen: ['UPSTREAM_HTTP: /models endpoint error', 'UPSTREAM_REFUSED: invalid API key'],
187
+ worksWith: ['validate_model', 'get_model_info'],
188
+ }),
189
+ get_model_info: buildToolDescription({
190
+ summary: 'Return pricing, context length, and modality architecture for one model id from the cached catalog.',
191
+ useWhen: [
192
+ 'You have a model id and need context window, pricing, or input/output modalities',
193
+ 'You are choosing between two known model slugs',
194
+ ],
195
+ notWhen: [
196
+ 'You only need true/false existence → validate_model (cheaper)',
197
+ 'You are browsing unknown models → search_models first',
198
+ ],
199
+ goodExamples: [
200
+ '`{ "model": "openai/gpt-4o" }`',
201
+ '`{ "model": "google/gemini-2.5-flash" }` before analyze_video',
202
+ ],
203
+ badExamples: [
204
+ '`{ "model": "" }` → INVALID_INPUT',
205
+ '`{ "name": "gpt-4o" }` → wrong key; use `model` with full slug `openai/gpt-4o`',
206
+ 'Calling repeatedly in a loop → cache is shared; call once per id',
207
+ ],
208
+ failsWhen: [
209
+ 'INVALID_INPUT: model not provided',
210
+ 'MODEL_NOT_FOUND: slug not in catalog',
211
+ 'UPSTREAM_HTTP: catalog refresh failed',
212
+ ],
213
+ worksWith: ['search_models', 'validate_model'],
214
+ }),
215
+ validate_model: buildToolDescription({
216
+ summary: 'Cheap boolean check: does this model id exist in the OpenRouter catalog? Uses the shared cache.',
217
+ useWhen: [
218
+ 'Pre-flight before chat_completion or generate_* to avoid MODEL_NOT_FOUND',
219
+ 'You only need `{ valid: true|false }`, not pricing or modalities',
220
+ ],
221
+ notWhen: [
222
+ 'You need pricing or context length → get_model_info',
223
+ 'You are discovering models → search_models',
224
+ ],
225
+ goodExamples: [
226
+ '`{ "model": "anthropic/claude-sonnet-4" }` → `{ "valid": true, "model": "..." }`',
227
+ '`{ "model": "fake/model" }` → `{ "valid": false }` (not an error)',
228
+ ],
229
+ badExamples: [
230
+ 'Treating `valid: false` as a tool error — it is a successful response',
231
+ 'Using validate_model to search partial names → use search_models with `query`',
232
+ ],
233
+ failsWhen: ['INVALID_INPUT: model not provided', 'UPSTREAM_HTTP: catalog refresh failed'],
234
+ worksWith: ['get_model_info', 'chat_completion'],
235
+ }),
236
+ generate_image: buildToolDescription({
237
+ summary: 'Generate an image from a text prompt. Optional `input_images` condition style/identity. ' +
238
+ 'Default model: google/gemini-2.5-flash-image.',
239
+ useWhen: [
240
+ 'You need a new image from a text prompt',
241
+ 'You have reference images for style or subject consistency',
242
+ ],
243
+ notWhen: [
244
+ 'You want to analyze an existing image → analyze_image',
245
+ 'You want video → generate_video or generate_video_from_image',
246
+ 'Prompt is empty or only whitespace',
247
+ ],
248
+ goodExamples: [
249
+ '`{ "prompt": "A watercolor fox in autumn leaves" }`',
250
+ '`{ "prompt": "Same character", "input_images": ["ref.png"], "aspect_ratio": "16:9" }`',
251
+ '`{ "prompt": "Logo", "save_path": "out/logo.png" }` inside output sandbox',
252
+ ],
253
+ badExamples: [
254
+ '`{ "prompt": "" }` → INVALID_INPUT',
255
+ '`{ "input_images": ["/etc/passwd"] }` → UNSAFE_PATH',
256
+ '`{ "aspect_ratio": "21:9" }` if not in allowed enum → INVALID_INPUT',
257
+ ],
258
+ failsWhen: [
259
+ 'INVALID_INPUT: empty prompt, bad aspect_ratio/image_size, unreadable reference',
260
+ 'UNSAFE_PATH: save_path or input_images escaped sandbox',
261
+ 'UPSTREAM_REFUSED: content policy or insufficient credits',
262
+ 'MODEL_NOT_FOUND: invalid model slug',
263
+ ],
264
+ worksWith: ['analyze_image', 'generate_video_from_image'],
265
+ }),
266
+ generate_audio: buildToolDescription({
267
+ summary: 'Generate speech or music from a text prompt. Output format is auto-detected; file extension auto-corrected on save.',
268
+ useWhen: [
269
+ 'You need TTS or audio generation from text',
270
+ 'Optional save_path is inside the output sandbox',
271
+ ],
272
+ notWhen: ['You want to transcribe existing audio → analyze_audio', 'Prompt is empty'],
273
+ goodExamples: [
274
+ '`{ "prompt": "Say hello world in a calm voice." }`',
275
+ '`{ "prompt": "Upbeat jingle", "save_path": "out/jingle.mp3" }`',
276
+ ],
277
+ badExamples: [
278
+ '`{ "text": "hello" }` → wrong key; use `prompt`',
279
+ '`{ "save_path": "../../../tmp/out.wav" }` → UNSAFE_PATH',
280
+ ],
281
+ failsWhen: [
282
+ 'INVALID_INPUT: prompt empty',
283
+ 'UNSAFE_PATH: save_path escaped sandbox',
284
+ 'UPSTREAM_REFUSED: content policy or credits',
285
+ ],
286
+ worksWith: ['analyze_audio'],
287
+ }),
288
+ generate_video: buildToolDescription({
289
+ summary: 'Generate video from a text prompt (optional first/last frame or reference images). Submits an async job, ' +
290
+ 'polls until `max_wait_ms`, downloads on completion. Emits MCP progress when client sends `progressToken`. ' +
291
+ 'Default model: google/veo-3.1.',
292
+ useWhen: [
293
+ 'You need text-to-video or frame-conditioned video',
294
+ 'You can wait for polling or resume later with get_video_status',
295
+ 'You need last_frame or multiple reference_images (not available on generate_video_from_image)',
296
+ ],
297
+ notWhen: [
298
+ 'You only have one image and simple image-to-video → generate_video_from_image (fewer params)',
299
+ 'Job already submitted → get_video_status with `video_id`',
300
+ 'You want to analyze existing video → analyze_video',
301
+ ],
302
+ goodExamples: [
303
+ '`{ "prompt": "Ocean waves at sunset, cinematic" }`',
304
+ 'Timeout resume: response has `_meta.code: JOB_STILL_RUNNING` and `_meta.video_id` → call `get_video_status`',
305
+ '`{ "prompt": "Morph", "first_frame_image": "a.jpg", "last_frame_image": "b.jpg" }`',
306
+ ],
307
+ badExamples: [
308
+ 'Treating JOB_STILL_RUNNING as failure — it is success with resume metadata',
309
+ '`{ "prompt": " " }` → INVALID_INPUT',
310
+ 'Polling get_video_status in the same turn without waiting → expect JOB_STILL_RUNNING again',
311
+ ],
312
+ failsWhen: [
313
+ 'INVALID_INPUT: empty prompt',
314
+ 'UNSAFE_PATH: save_path or image paths escaped sandbox',
315
+ 'UPSTREAM_REFUSED: policy, credits, or bad request',
316
+ 'JOB_FAILED: provider marked job failed',
317
+ ],
318
+ worksWith: ['get_video_status', 'generate_video_from_image'],
319
+ }),
320
+ generate_video_from_image: buildToolDescription({
321
+ summary: 'Narrow image-to-video wrapper: one `image` (first frame) + `prompt`. Fewer parameters → higher tool-call accuracy. ' +
322
+ 'For last-frame or reference images use generate_video.',
323
+ useWhen: [
324
+ 'Single reference image + motion prompt is enough',
325
+ 'You want the smallest argument surface for image-to-video',
326
+ ],
327
+ notWhen: [
328
+ 'You need last_frame_image or reference_images[] → generate_video',
329
+ 'Checking job status → get_video_status',
330
+ ],
331
+ goodExamples: [
332
+ '`{ "image": "start.png", "prompt": "Camera slowly zooms in" }`',
333
+ 'On timeout: same JOB_STILL_RUNNING + video_id resume as generate_video',
334
+ ],
335
+ badExamples: [
336
+ '`{ "first_frame_image": "x.png" }` → wrong key; use `image`',
337
+ 'Passing video_id here → use get_video_status',
338
+ ],
339
+ failsWhen: [
340
+ 'INVALID_INPUT: image or prompt missing',
341
+ 'UNSAFE_PATH: image path escaped sandbox',
342
+ 'UPSTREAM_REFUSED / JOB_FAILED: same as generate_video',
343
+ ],
344
+ worksWith: ['generate_video', 'get_video_status'],
345
+ }),
346
+ get_video_status: buildToolDescription({
347
+ summary: 'Poll an async video job by id. Downloads and optionally saves when complete. ' +
348
+ 'Still running → success with `_meta.code: JOB_STILL_RUNNING` (not an error).',
349
+ useWhen: [
350
+ 'generate_video returned JOB_STILL_RUNNING or you have a video_id from a prior call',
351
+ 'You want to check progress without resubmitting',
352
+ ],
353
+ notWhen: [
354
+ 'Starting a new generation → generate_video or generate_video_from_image',
355
+ 'You do not have a video_id yet',
356
+ ],
357
+ goodExamples: [
358
+ '`{ "video_id": "vid_abc123" }`',
359
+ '`{ "video_id": "vid_abc123", "save_path": "out/clip.mp4" }`',
360
+ 'Repeat until status completes or you accept partial progress from `_meta.progress`',
361
+ ],
362
+ badExamples: [
363
+ '`{ "id": "vid_abc" }` → wrong key; use `video_id`',
364
+ 'Expecting instant completion on first poll for long jobs',
365
+ 'Treating JOB_STILL_RUNNING as tool failure',
366
+ ],
367
+ failsWhen: [
368
+ 'INVALID_INPUT: video_id missing',
369
+ 'UNSAFE_PATH: save_path escaped sandbox',
370
+ 'JOB_FAILED: provider marked job failed',
371
+ ],
372
+ worksWith: ['generate_video', 'generate_video_from_image'],
373
+ }),
374
+ rerank_documents: buildToolDescription({
375
+ summary: 'Re-order documents by relevance to a query using an OpenRouter reranker. Default: cohere/rerank-english-v3.0.',
376
+ useWhen: [
377
+ 'You have a query and a list of text snippets to sort by relevance',
378
+ 'You will feed top results into chat_completion for grounded answers',
379
+ ],
380
+ notWhen: [
381
+ 'You need to fetch documents from the web → chat_completion with online or external retrieval first',
382
+ 'documents is empty or contains non-strings',
383
+ ],
384
+ goodExamples: [
385
+ '`{ "query": "battery life", "documents": ["Doc A text...", "Doc B text..."] }`',
386
+ '`{ "query": "...", "documents": [...], "model": "cohere/rerank-english-v3.0" }`',
387
+ ],
388
+ badExamples: [
389
+ '`{ "documents": [] }` → INVALID_INPUT',
390
+ '`{ "query": "x", "documents": [{ "text": "y" }] }` → elements must be strings',
391
+ 'Using rerank output as model messages without extracting text fields',
392
+ ],
393
+ failsWhen: [
394
+ 'INVALID_INPUT: query missing, documents empty, or non-string elements',
395
+ 'MODEL_NOT_FOUND: reranker slug invalid',
396
+ 'UPSTREAM_HTTP: provider error',
397
+ ],
398
+ worksWith: ['search_models', 'chat_completion'],
399
+ }),
400
+ health_check: buildToolDescription({
401
+ summary: 'Verify API key, OpenRouter reachability, cached model count, and server/protocol versions. No arguments.',
402
+ useWhen: [
403
+ 'Startup / ops probe before other tools',
404
+ 'You need `{ ok, api_key_valid }` without triggering generation costs',
405
+ ],
406
+ notWhen: [
407
+ 'You need to test a specific model quality → use chat_completion with a tiny prompt',
408
+ 'You expect isError on bad API key — this tool always returns structured payload',
409
+ ],
410
+ goodExamples: [
411
+ '`{}` — empty args',
412
+ 'Branch on `structuredContent.api_key_valid === false` to prompt re-auth',
413
+ ],
414
+ badExamples: [
415
+ 'Passing model or prompt — ignored; not a chat tool',
416
+ 'Expecting isError: true on failure — check `ok` field instead',
417
+ ],
418
+ failsWhen: [
419
+ 'Never returns isError — always `{ ok, api_key_valid, ... }` for programmatic branching',
420
+ ],
421
+ worksWith: ['every other tool (run once at startup)'],
422
+ }),
423
+ };
@@ -1,9 +1,10 @@
1
1
  import { prepareAudioData } from './audio-utils.js';
2
+ import { UnsafeOutputPathError } from './path-safety.js';
2
3
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
3
4
  import { SERVER_VERSION } from '../version.js';
4
5
  import { classifyUpstreamError } from './openrouter-errors.js';
5
6
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
6
- import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
7
+ import { buildCacheHeaders, extractCacheMeta } from './cache.js';
7
8
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
8
9
  const DEFAULT_MODEL = 'google/gemini-2.5-flash';
9
10
  export async function handleAnalyzeAudio(request, openai, defaultModel) {
@@ -17,6 +18,9 @@ export async function handleAnalyzeAudio(request, openai, defaultModel) {
17
18
  audioData = await prepareAudioData(audio_path);
18
19
  }
19
20
  catch (err) {
21
+ if (err instanceof UnsafeOutputPathError) {
22
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
23
+ }
20
24
  const msg = err instanceof Error ? err.message : String(err);
21
25
  if (msg.includes('Blocked host'))
22
26
  return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
@@ -1,9 +1,10 @@
1
1
  import { prepareImageUrl } from './image-utils.js';
2
+ import { UnsafeOutputPathError } from './path-safety.js';
2
3
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
3
4
  import { SERVER_VERSION } from '../version.js';
4
5
  import { classifyUpstreamError } from './openrouter-errors.js';
5
6
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
6
- import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
7
+ import { buildCacheHeaders, extractCacheMeta } from './cache.js';
7
8
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
8
9
  const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
9
10
  export async function handleAnalyzeImage(request, openai, defaultModel) {
@@ -17,6 +18,9 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
17
18
  imageUrl = await prepareImageUrl(image_path);
18
19
  }
19
20
  catch (err) {
21
+ if (err instanceof UnsafeOutputPathError) {
22
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
23
+ }
20
24
  const msg = err instanceof Error ? err.message : String(err);
21
25
  if (msg.includes('Blocked host'))
22
26
  return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
@@ -44,10 +48,7 @@ export async function handleAnalyzeImage(request, openai, defaultModel) {
44
48
  messages: [
45
49
  {
46
50
  role: 'user',
47
- content: [
48
- { type: 'text', text: question || "What's in this image?" },
49
- imageBlock,
50
- ],
51
+ content: [{ type: 'text', text: question || "What's in this image?" }, imageBlock],
51
52
  },
52
53
  ],
53
54
  }, requestOpts);
@@ -1,10 +1,11 @@
1
1
  import { prepareVideoData } from './video-utils.js';
2
+ import { UnsafeOutputPathError } from './path-safety.js';
2
3
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
3
4
  import { SERVER_VERSION } from '../version.js';
4
5
  import { logger } from '../logger.js';
5
6
  import { classifyUpstreamError } from './openrouter-errors.js';
6
7
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
7
- import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
8
+ import { buildCacheHeaders, extractCacheMeta } from './cache.js';
8
9
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
9
10
  /**
10
11
  * Default model — `google/gemini-2.5-flash` has the widest video-input
@@ -18,15 +19,15 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
18
19
  if (!video_path) {
19
20
  return toolError(ErrorCode.INVALID_INPUT, 'video_path is required.');
20
21
  }
21
- const pickedModel = model ||
22
- process.env.OPENROUTER_DEFAULT_VIDEO_MODEL ||
23
- defaultModel ||
24
- FALLBACK_DEFAULT_MODEL;
22
+ const pickedModel = model || process.env.OPENROUTER_DEFAULT_VIDEO_MODEL || defaultModel || FALLBACK_DEFAULT_MODEL;
25
23
  let videoData;
26
24
  try {
27
25
  videoData = await prepareVideoData(video_path);
28
26
  }
29
27
  catch (err) {
28
+ if (err instanceof UnsafeOutputPathError) {
29
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
30
+ }
30
31
  const msg = err instanceof Error ? err.message : String(err);
31
32
  if (msg.includes('Blocked host')) {
32
33
  return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
@@ -5,6 +5,7 @@
5
5
  import path from 'path';
6
6
  import { promises as fs } from 'fs';
7
7
  import { readEnvInt, fetchHttpResource, parseBase64DataUrl } from './fetch-utils.js';
8
+ import { resolveSafeInputPath } from './path-safety.js';
8
9
  // Re-export for tests
9
10
  export { isBlockedIPv4, assertUrlSafeForFetch } from './fetch-utils.js';
10
11
  const DEFAULT_FETCH_TIMEOUT_MS = 30_000;
@@ -117,10 +118,11 @@ export async function prepareAudioData(source) {
117
118
  return { data: buffer.toString('base64'), format };
118
119
  }
119
120
  // --- local file ---
120
- const format = getAudioFormat(source);
121
+ const safe = await resolveSafeInputPath(source);
122
+ const format = getAudioFormat(safe);
121
123
  if (!format) {
122
124
  throw new Error(`Unsupported audio format for file: ${source}. Supported: ${SUPPORTED_AUDIO_FORMATS.join(', ')}`);
123
125
  }
124
- const buffer = await fs.readFile(source);
126
+ const buffer = await fs.readFile(safe);
125
127
  return { data: buffer.toString('base64'), format };
126
128
  }
@@ -3,7 +3,7 @@ import { SERVER_VERSION } from '../version.js';
3
3
  import { classifyUpstreamError } from './openrouter-errors.js';
4
4
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
5
5
  import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
6
- import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
6
+ import { buildCacheHeaders, extractCacheMeta } from './cache.js';
7
7
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
8
8
  const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
9
9
  function readIncludeReasoningDefault() {
@@ -151,11 +151,25 @@ export function isBlockedIPv6(ip) {
151
151
  return false;
152
152
  const [g0, g1, g2, g3, g4, g5, g6, g7] = groups;
153
153
  // :: (unspecified)
154
- if (g0 === 0 && g1 === 0 && g2 === 0 && g3 === 0 && g4 === 0 && g5 === 0 && g6 === 0 && g7 === 0) {
154
+ if (g0 === 0 &&
155
+ g1 === 0 &&
156
+ g2 === 0 &&
157
+ g3 === 0 &&
158
+ g4 === 0 &&
159
+ g5 === 0 &&
160
+ g6 === 0 &&
161
+ g7 === 0) {
155
162
  return true;
156
163
  }
157
164
  // ::1 (loopback)
158
- if (g0 === 0 && g1 === 0 && g2 === 0 && g3 === 0 && g4 === 0 && g5 === 0 && g6 === 0 && g7 === 1) {
165
+ if (g0 === 0 &&
166
+ g1 === 0 &&
167
+ g2 === 0 &&
168
+ g3 === 0 &&
169
+ g4 === 0 &&
170
+ g5 === 0 &&
171
+ g6 === 0 &&
172
+ g7 === 1) {
159
173
  return true;
160
174
  }
161
175
  // ::ffff:0:0/96 — IPv4-mapped. Re-check the embedded IPv4.
@@ -95,9 +95,7 @@ export async function handleGenerateAudio(request, openai) {
95
95
  logger.audit('generate_audio.start', {
96
96
  model: model || DEFAULT_MODEL,
97
97
  voice: voice?.trim() || DEFAULT_VOICE,
98
- format: VALID_FORMATS.includes(format ?? '')
99
- ? format
100
- : DEFAULT_FORMAT,
98
+ format: VALID_FORMATS.includes(format ?? '') ? format : DEFAULT_FORMAT,
101
99
  prompt_preview: prompt.slice(0, 80),
102
100
  save_path: save_path ? 'provided' : 'none',
103
101
  });
@@ -156,7 +154,7 @@ export async function handleGenerateAudio(request, openai) {
156
154
  const detected = detectAudioFormat(audioBuffer);
157
155
  // Always wrap raw PCM in WAV so it's playable
158
156
  if (detected.ext === 'pcm') {
159
- audioBuffer = wrapPcmInWav(audioBuffer);
157
+ audioBuffer = Buffer.from(wrapPcmInWav(audioBuffer));
160
158
  detected.ext = 'wav';
161
159
  detected.mimeType = 'audio/wav';
162
160
  }
@@ -0,0 +1,3 @@
1
+ import OpenAI from 'openai';
2
+ export declare function resolveInputImage(ref: string): Promise<string>;
3
+ export declare function buildUserContent(prompt: string, inputImages?: string[]): Promise<string | OpenAI.Chat.Completions.ChatCompletionContentPart[]>;
@@ -0,0 +1,38 @@
1
+ import { promises as fs } from 'fs';
2
+ import path from 'node:path';
3
+ import { resolveSafeInputPath } from './path-safety.js';
4
+ import { mimeFromExtension } from './image-utils.js';
5
+ export async function resolveInputImage(ref) {
6
+ const trimmed = ref.trim();
7
+ if (!trimmed)
8
+ throw new Error('empty input_images entry');
9
+ if (trimmed.startsWith('data:'))
10
+ return trimmed;
11
+ if (/^https?:\/\//i.test(trimmed))
12
+ return trimmed;
13
+ const abs = await resolveSafeInputPath(trimmed);
14
+ const buf = await fs.readFile(abs);
15
+ const mime = mimeFromExtension(path.extname(abs)) || 'image/png';
16
+ return `data:${mime};base64,${buf.toString('base64')}`;
17
+ }
18
+ export async function buildUserContent(prompt, inputImages) {
19
+ if (!inputImages?.length) {
20
+ return `Generate an image: ${prompt}`;
21
+ }
22
+ const parts = [
23
+ {
24
+ type: 'text',
25
+ text: `Generate an image based on this prompt, using the following reference image(s) ` +
26
+ `for visual consistency. Match the appearance, identity, and style of the references ` +
27
+ `closely; do not alter them.\n\nPrompt: ${prompt}`,
28
+ },
29
+ ];
30
+ const urls = await Promise.all(inputImages.map((ref) => resolveInputImage(ref)));
31
+ for (const url of urls) {
32
+ parts.push({
33
+ type: 'image_url',
34
+ image_url: { url, detail: 'high' },
35
+ });
36
+ }
37
+ return parts;
38
+ }