@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +367 -283
  2. package/dist/index.js +1 -1
  3. package/dist/model-cache.d.ts +22 -12
  4. package/dist/model-cache.js +58 -21
  5. package/dist/openrouter-api.d.ts +45 -0
  6. package/dist/openrouter-api.js +50 -0
  7. package/dist/tool-descriptions.d.ts +19 -0
  8. package/dist/tool-descriptions.js +573 -0
  9. package/dist/tool-handlers/analyze-audio.js +5 -1
  10. package/dist/tool-handlers/analyze-image.js +6 -5
  11. package/dist/tool-handlers/analyze-video.js +6 -5
  12. package/dist/tool-handlers/async-chat.d.ts +51 -0
  13. package/dist/tool-handlers/async-chat.js +216 -0
  14. package/dist/tool-handlers/audio-utils.js +4 -2
  15. package/dist/tool-handlers/chat-completion.js +1 -1
  16. package/dist/tool-handlers/fetch-utils.js +16 -2
  17. package/dist/tool-handlers/generate-audio.js +2 -4
  18. package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
  19. package/dist/tool-handlers/generate-image-dedicated.js +176 -0
  20. package/dist/tool-handlers/generate-image-input.d.ts +3 -0
  21. package/dist/tool-handlers/generate-image-input.js +38 -0
  22. package/dist/tool-handlers/generate-image.d.ts +13 -51
  23. package/dist/tool-handlers/generate-image.js +32 -119
  24. package/dist/tool-handlers/generate-video.d.ts +2 -2
  25. package/dist/tool-handlers/generate-video.js +78 -30
  26. package/dist/tool-handlers/image-utils.d.ts +1 -0
  27. package/dist/tool-handlers/image-utils.js +26 -16
  28. package/dist/tool-handlers/openrouter-errors.js +6 -2
  29. package/dist/tool-handlers/path-safety.js +32 -5
  30. package/dist/tool-handlers/provider-routing.js +7 -2
  31. package/dist/tool-handlers/rerank.js +2 -5
  32. package/dist/tool-handlers/search-models.d.ts +2 -2
  33. package/dist/tool-handlers/search-models.js +2 -6
  34. package/dist/tool-handlers/speech-to-text.d.ts +20 -0
  35. package/dist/tool-handlers/speech-to-text.js +140 -0
  36. package/dist/tool-handlers/structured-output.d.ts +8 -0
  37. package/dist/tool-handlers/structured-output.js +11 -0
  38. package/dist/tool-handlers/text-to-speech.d.ts +29 -0
  39. package/dist/tool-handlers/text-to-speech.js +105 -0
  40. package/dist/tool-handlers/video-utils.js +6 -9
  41. package/dist/tool-handlers.js +253 -125
  42. package/dist/version.d.ts +1 -1
  43. package/dist/version.js +1 -1
  44. package/package.json +27 -15
@@ -3,48 +3,10 @@ export interface GenerateImageToolRequest {
3
3
  prompt: string;
4
4
  model?: string;
5
5
  save_path?: string;
6
- /**
7
- * Output aspect ratio. Passed through as `image_config.aspect_ratio`.
8
- * Supported by OpenRouter image models (e.g. `1:1`, `16:9`, `9:16`,
9
- * `4:3`, `3:4`, `21:9`). Model-dependent. Unsupported values fall back
10
- * to the model's default. See
11
- * https://openrouter.ai/docs/guides/overview/multimodal/image-generation
12
- */
13
6
  aspect_ratio?: string;
14
- /**
15
- * Output image resolution bucket. Passed through as
16
- * `image_config.image_size`. Typical values: `0.5K`, `1K` (default),
17
- * `2K`, `4K`. Model-dependent.
18
- */
19
7
  image_size?: string;
20
- /**
21
- * Upper bound on the completion budget. Without this OpenRouter
22
- * reserves the model's full context window (~29k for Gemini
23
- * image models), which can trigger a 402 on low-credit accounts even
24
- * though the actual generation uses far fewer tokens. 4096 is plenty
25
- * for the image payload + any caption.
26
- */
27
8
  max_tokens?: number;
28
- /**
29
- * Optional reference images. Each entry is one of:
30
- * - a `data:image/...;base64,...` URL,
31
- * - an `http(s)://` URL (OpenRouter fetches it),
32
- * - a local file path (sandboxed to `OPENROUTER_INPUT_DIR` /
33
- * `OPENROUTER_OUTPUT_DIR` / cwd, read + base64-encoded by the server).
34
- *
35
- * When provided, the user message becomes multimodal: a text prompt
36
- * plus one `image_url` block per reference, in array order. Enables
37
- * character / style consistency and image-to-image refinement on
38
- * chat-image models that accept image inputs (Gemini Nano Banana,
39
- * `openai/gpt-5.4-image-2`).
40
- */
41
9
  input_images?: string[];
42
- /**
43
- * Override the default `modalities: ["image","text"]` sent to
44
- * OpenRouter. Most callers should leave this unset. Provide e.g.
45
- * `["text"]` to suppress image output for inspection / captioning,
46
- * or other shapes for future model variants.
47
- */
48
10
  modalities?: string[];
49
11
  }
50
12
  export declare function handleGenerateImage(request: {
@@ -64,11 +26,16 @@ export declare function handleGenerateImage(request: {
64
26
  text?: undefined;
65
27
  })[];
66
28
  _meta: {
67
- usage?: {
29
+ usage: {
68
30
  prompt_tokens: number;
69
31
  completion_tokens: number;
70
32
  total_tokens: number;
71
- } | undefined;
33
+ };
34
+ server_version: string;
35
+ save_path: string;
36
+ mime: string;
37
+ } | {
38
+ usage?: undefined;
72
39
  server_version: string;
73
40
  save_path: string;
74
41
  mime: string;
@@ -80,21 +47,16 @@ export declare function handleGenerateImage(request: {
80
47
  data: string;
81
48
  }[];
82
49
  _meta: {
83
- usage?: {
50
+ usage: {
84
51
  prompt_tokens: number;
85
52
  completion_tokens: number;
86
53
  total_tokens: number;
87
- } | undefined;
54
+ };
55
+ server_version: string;
56
+ mime: string;
57
+ } | {
58
+ usage?: undefined;
88
59
  server_version: string;
89
60
  mime: string;
90
61
  };
91
62
  }>;
92
- /**
93
- * Resolve a caller-supplied input image into a URL the chat-completions
94
- * API accepts. Local file paths are sandboxed via
95
- * `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
96
- * `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
97
- */
98
- export declare function resolveInputImage(ref: string): Promise<string>;
99
- export declare function mimeFromExt(ext: string): string | null;
100
- export declare function buildUserContent(prompt: string, inputImages?: string[]): Promise<string | OpenAI.Chat.Completions.ChatCompletionContentPart[]>;
@@ -1,15 +1,12 @@
1
1
  import { promises as fs } from 'fs';
2
- import path from 'node:path';
3
- import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
2
+ import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
4
3
  import { parseBase64DataUrl } from './fetch-utils.js';
4
+ import { buildUserContent } from './generate-image-input.js';
5
5
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
6
6
  import { SERVER_VERSION } from '../version.js';
7
7
  import { logger } from '../logger.js';
8
8
  import { classifyUpstreamError } from './openrouter-errors.js';
9
9
  const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
10
- // OpenRouter-documented aspect ratios (standard + extended). Extended are
11
- // only honored by models that support them (e.g. gemini-3.1-flash-image),
12
- // others fall back to the model's default.
13
10
  const VALID_ASPECT_RATIOS = new Set([
14
11
  '1:1',
15
12
  '2:3',
@@ -32,9 +29,6 @@ export async function handleGenerateImage(request, openai) {
32
29
  if (!prompt?.trim()) {
33
30
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
34
31
  }
35
- // Audit entry. Bypasses the normal log level so operators always see a
36
- // record of cost-incurring operations. Prompt preview is hard-capped
37
- // at 80 chars to avoid PII spillage in log aggregators.
38
32
  logger.audit('generate_image.start', {
39
33
  model: model || DEFAULT_MODEL,
40
34
  prompt_preview: prompt.slice(0, 80),
@@ -43,15 +37,12 @@ export async function handleGenerateImage(request, openai) {
43
37
  save_path: save_path ? 'provided' : 'none',
44
38
  input_images_count: input_images?.length ?? 0,
45
39
  });
46
- // Validate optional shape fields early so callers get a clear error
47
- // instead of a cryptic upstream 400.
48
40
  if (aspect_ratio !== undefined && !VALID_ASPECT_RATIOS.has(aspect_ratio)) {
49
- return toolError(ErrorCode.INVALID_INPUT, `aspect_ratio '${aspect_ratio}' is not supported. Valid values: ${[...VALID_ASPECT_RATIOS].join(', ')}.`);
41
+ return invalidEnumError('aspect_ratio', aspect_ratio, VALID_ASPECT_RATIOS);
50
42
  }
51
43
  if (image_size !== undefined && !VALID_IMAGE_SIZES.has(image_size)) {
52
- return toolError(ErrorCode.INVALID_INPUT, `image_size '${image_size}' is not supported. Valid values: ${[...VALID_IMAGE_SIZES].join(', ')}.`);
44
+ return invalidEnumError('image_size', image_size, VALID_IMAGE_SIZES);
53
45
  }
54
- // Fail-fast on unsafe paths BEFORE spending tokens.
55
46
  let safePathResolved = null;
56
47
  if (save_path) {
57
48
  try {
@@ -64,9 +55,6 @@ export async function handleGenerateImage(request, openai) {
64
55
  return toolErrorFrom(ErrorCode.INTERNAL, err);
65
56
  }
66
57
  }
67
- // Build the user message. With no `input_images`, this is the original
68
- // string content; with refs, it becomes a multimodal
69
- // ChatCompletionContentPart[] (text preamble + one image_url per ref).
70
58
  let content;
71
59
  try {
72
60
  content = await buildUserContent(prompt, input_images);
@@ -77,14 +65,6 @@ export async function handleGenerateImage(request, openai) {
77
65
  }
78
66
  return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_images');
79
67
  }
80
- // Assemble the request body. OpenRouter's image-generation guide
81
- // requires:
82
- // - `modalities: ["image", "text"]` so multimodal models (like
83
- // Gemini) know to emit an image, not just text. Caller can
84
- // override via the `modalities` field.
85
- // - `image_config.{aspect_ratio, image_size}` for shape control.
86
- // The OpenAI SDK doesn't type these fields, but passes unknown members
87
- // through to the server, so we attach them via a typed cast.
88
68
  const imageConfig = {};
89
69
  if (aspect_ratio)
90
70
  imageConfig.aspect_ratio = aspect_ratio;
@@ -93,7 +73,7 @@ export async function handleGenerateImage(request, openai) {
93
73
  const body = {
94
74
  model: model || DEFAULT_MODEL,
95
75
  messages: [{ role: 'user', content }],
96
- modalities: modalities && modalities.length ? modalities : ['image', 'text'],
76
+ modalities: modalities?.length ? modalities : ['image', 'text'],
97
77
  };
98
78
  if (Object.keys(imageConfig).length > 0)
99
79
  body.image_config = imageConfig;
@@ -101,10 +81,6 @@ export async function handleGenerateImage(request, openai) {
101
81
  body.max_tokens = max_tokens;
102
82
  let completion;
103
83
  try {
104
- // OpenRouter-specific `image_config` isn't in the OpenAI SDK's typings,
105
- // but the SDK passes unknown fields straight through to the server.
106
- // We never pass `stream: true`, so the response is always
107
- // ChatCompletion.
108
84
  completion = (await openai.chat.completions.create(body));
109
85
  }
110
86
  catch (err) {
@@ -116,8 +92,6 @@ export async function handleGenerateImage(request, openai) {
116
92
  }
117
93
  const base64 = extractBase64(message);
118
94
  if (!base64) {
119
- // Model talked but did not emit an image. Surface this as a distinct
120
- // condition so callers don't treat chatter as a successful image.
121
95
  const messageContent = message.content;
122
96
  const text = typeof messageContent === 'string' ? messageContent : JSON.stringify(messageContent);
123
97
  return toolError(ErrorCode.UPSTREAM_REFUSED, `Model returned no image. Text response: ${text.slice(0, 300)}`, {
@@ -127,113 +101,56 @@ export async function handleGenerateImage(request, openai) {
127
101
  }
128
102
  if (safePathResolved) {
129
103
  try {
130
- await fs.writeFile(safePathResolved, Buffer.from(base64.data, 'base64'));
104
+ await fs.writeFile(safePathResolved, base64.data, { encoding: 'base64' });
131
105
  }
132
106
  catch (err) {
133
107
  return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
134
108
  }
135
- const usage = completion.usage;
109
+ }
110
+ return buildImageSuccessResult(base64, completion.usage, safePathResolved ?? undefined);
111
+ }
112
+ function invalidEnumError(field, value, allowed) {
113
+ return toolError(ErrorCode.INVALID_INPUT, `${field} '${value}' is not supported. Valid values: ${[...allowed].join(', ')}.`);
114
+ }
115
+ function buildImageSuccessResult(base64, usage, savePath) {
116
+ const usageMeta = usage
117
+ ? {
118
+ usage: {
119
+ prompt_tokens: usage.prompt_tokens,
120
+ completion_tokens: usage.completion_tokens,
121
+ total_tokens: usage.total_tokens,
122
+ },
123
+ }
124
+ : {};
125
+ if (savePath) {
136
126
  return {
137
127
  content: [
138
- { type: 'text', text: `Image saved to: ${safePathResolved}` },
128
+ { type: 'text', text: `Image saved to: ${savePath}` },
139
129
  { type: 'image', mimeType: base64.mime, data: base64.data },
140
130
  ],
141
131
  _meta: {
142
132
  server_version: SERVER_VERSION,
143
- save_path: safePathResolved,
133
+ save_path: savePath,
144
134
  mime: base64.mime,
145
- ...(usage
146
- ? {
147
- usage: {
148
- prompt_tokens: usage.prompt_tokens,
149
- completion_tokens: usage.completion_tokens,
150
- total_tokens: usage.total_tokens,
151
- },
152
- }
153
- : {}),
135
+ ...usageMeta,
154
136
  },
155
137
  };
156
138
  }
157
- const usage = completion.usage;
158
139
  return {
159
140
  content: [{ type: 'image', mimeType: base64.mime, data: base64.data }],
160
141
  _meta: {
161
142
  server_version: SERVER_VERSION,
162
143
  mime: base64.mime,
163
- ...(usage
164
- ? {
165
- usage: {
166
- prompt_tokens: usage.prompt_tokens,
167
- completion_tokens: usage.completion_tokens,
168
- total_tokens: usage.total_tokens,
169
- },
170
- }
171
- : {}),
144
+ ...usageMeta,
172
145
  },
173
146
  };
174
147
  }
175
- /**
176
- * Resolve a caller-supplied input image into a URL the chat-completions
177
- * API accepts. Local file paths are sandboxed via
178
- * `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
179
- * `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
180
- */
181
- export async function resolveInputImage(ref) {
182
- const trimmed = ref.trim();
183
- if (!trimmed)
184
- throw new Error('empty input_images entry');
185
- if (trimmed.startsWith('data:'))
186
- return trimmed;
187
- if (/^https?:\/\//i.test(trimmed))
188
- return trimmed;
189
- const abs = await resolveSafeInputPath(trimmed);
190
- const buf = await fs.readFile(abs);
191
- const mime = mimeFromExt(path.extname(abs)) || 'image/png';
192
- return `data:${mime};base64,${buf.toString('base64')}`;
193
- }
194
- export function mimeFromExt(ext) {
195
- const e = ext.toLowerCase().replace(/^\./, '');
196
- switch (e) {
197
- case 'png':
198
- return 'image/png';
199
- case 'jpg':
200
- case 'jpeg':
201
- return 'image/jpeg';
202
- case 'webp':
203
- return 'image/webp';
204
- case 'gif':
205
- return 'image/gif';
206
- default:
207
- return null;
208
- }
209
- }
210
- export async function buildUserContent(prompt, inputImages) {
211
- if (!inputImages?.length) {
212
- return `Generate an image: ${prompt}`;
213
- }
214
- const parts = [
215
- {
216
- type: 'text',
217
- text: `Generate an image based on this prompt, using the following reference image(s) ` +
218
- `for visual consistency. Match the appearance, identity, and style of the references ` +
219
- `closely; do not alter them.\n\nPrompt: ${prompt}`,
220
- },
221
- ];
222
- for (const ref of inputImages) {
223
- const url = await resolveInputImage(ref);
224
- parts.push({
225
- type: 'image_url',
226
- image_url: { url, detail: 'high' },
227
- });
228
- }
229
- return parts;
230
- }
231
148
  function extractBase64(message) {
232
149
  const images = message.images;
233
150
  if (Array.isArray(images) && images.length) {
234
151
  for (const img of images) {
235
152
  const imageUrl = img.image_url;
236
- const result = parseDataUrl(imageUrl?.url || img.url);
153
+ const result = dataUrlToBase64(imageUrl?.url || img.url);
237
154
  if (result)
238
155
  return result;
239
156
  }
@@ -243,7 +160,7 @@ function extractBase64(message) {
243
160
  const iu = part.image_url;
244
161
  const url = iu?.url || part.url;
245
162
  if (url) {
246
- const r = parseDataUrl(url);
163
+ const r = dataUrlToBase64(url);
247
164
  if (r)
248
165
  return r;
249
166
  }
@@ -257,24 +174,20 @@ function extractBase64(message) {
257
174
  }
258
175
  }
259
176
  if (typeof message.content === 'string') {
260
- // Scan the string for an embedded data URL. We deliberately don't use
261
- // a single regex here because data URLs may carry MIME parameters
262
- // (e.g. `data:image/png;charset=binary;base64,...`) which trips the
263
- // naive `data:([^;]+);base64,(.+)` form.
177
+ // Data URLs may carry MIME parameters (e.g. charset=binary) that break naive regexes.
264
178
  const start = message.content.indexOf('data:image/');
265
179
  if (start >= 0) {
266
- // Find the end of the data URL: a whitespace or closing quote/paren.
267
180
  const tail = message.content.slice(start);
268
181
  const end = tail.search(/[\s)"']/);
269
182
  const url = end === -1 ? tail : tail.slice(0, end);
270
- const parsed = parseDataUrl(url);
183
+ const parsed = dataUrlToBase64(url);
271
184
  if (parsed)
272
185
  return parsed;
273
186
  }
274
187
  }
275
188
  return null;
276
189
  }
277
- function parseDataUrl(url) {
190
+ function dataUrlToBase64(url) {
278
191
  if (!url?.startsWith('data:'))
279
192
  return null;
280
193
  const parsed = parseBase64DataUrl(url);
@@ -34,7 +34,7 @@ export declare function handleGenerateVideo(request: {
34
34
  };
35
35
  }, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
36
36
  content: {
37
- type: "text";
37
+ type: string;
38
38
  text: string;
39
39
  }[];
40
40
  isError: false;
@@ -97,7 +97,7 @@ export declare function handleGenerateVideoFromImage(request: {
97
97
  };
98
98
  }, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
99
99
  content: {
100
- type: "text";
100
+ type: string;
101
101
  text: string;
102
102
  }[];
103
103
  isError: false;
@@ -11,6 +11,34 @@ const DEFAULT_POLL_INTERVAL_MS = 15_000;
11
11
  const DEFAULT_MAX_WAIT_MS = 10 * 60_000;
12
12
  const MIN_POLL_INTERVAL_MS = 50; // just to avoid a 0ms busy-loop if a caller omits
13
13
  const INLINE_RETURN_CEILING_BYTES = 10 * 1024 * 1024;
14
+ /** Models deprecated by OpenAI — removal date: 2026-09-24. */
15
+ const SORA_DEPRECATED_MODELS = new Set([
16
+ 'openai/sora-2',
17
+ 'openai/sora-2-pro',
18
+ 'openai/sora-2-2025-10-06',
19
+ 'openai/sora-2-2025-12-08',
20
+ 'openai/sora-2-pro-2025-10-06',
21
+ ]);
22
+ const SORA_ALTERNATIVES = [
23
+ 'google/veo-3.1 (recommended — fast, audio support)',
24
+ 'google/veo-3.1-fast (budget-friendly)',
25
+ 'bytedance/seedance-2.0 (high quality)',
26
+ 'bytedance/seedance-2.0-fast (fast turnaround)',
27
+ 'alibaba/wan-2.7 (good for artistic styles)',
28
+ ];
29
+ /**
30
+ * Check if the model is a deprecated Sora model and return a warning string,
31
+ * or null if no deprecation applies.
32
+ */
33
+ function checkSoraDeprecation(model) {
34
+ const normalized = model.toLowerCase().trim();
35
+ if (!SORA_DEPRECATED_MODELS.has(normalized) && !normalized.startsWith('openai/sora')) {
36
+ return null;
37
+ }
38
+ return (`⚠️ DEPRECATION WARNING: ${model} is deprecated by OpenAI and will be removed from the API on September 24, 2026. ` +
39
+ `Your request will still be attempted, but may fail. Recommended alternatives:\n` +
40
+ SORA_ALTERNATIVES.map((a) => ` • ${a}`).join('\n'));
41
+ }
14
42
  function getMaxInlineBytes() {
15
43
  return readEnvInt('OPENROUTER_VIDEO_INLINE_MAX_BYTES', INLINE_RETURN_CEILING_BYTES, 4096);
16
44
  }
@@ -88,34 +116,45 @@ function buildRequestBody(args, model) {
88
116
  return body;
89
117
  }
90
118
  async function attachFrameImages(args, body) {
91
- const frameImages = [];
119
+ const frameTasks = [];
92
120
  if (args.first_frame_image) {
93
- const img = await prepareImageInput(args.first_frame_image);
94
- if (img) {
95
- frameImages.push({
96
- frame_type: 'first_frame',
97
- image: { url: `data:${img.mime};base64,${img.data}` },
98
- });
99
- }
121
+ frameTasks.push(prepareImageInput(args.first_frame_image).then((img) => img
122
+ ? {
123
+ kind: 'frame',
124
+ entry: {
125
+ type: 'image_url',
126
+ image_url: { url: `data:${img.mime};base64,${img.data}` },
127
+ frame_type: 'first_frame',
128
+ },
129
+ }
130
+ : null));
100
131
  }
101
132
  if (args.last_frame_image) {
102
- const img = await prepareImageInput(args.last_frame_image);
103
- if (img) {
104
- frameImages.push({
105
- frame_type: 'last_frame',
106
- image: { url: `data:${img.mime};base64,${img.data}` },
107
- });
108
- }
133
+ frameTasks.push(prepareImageInput(args.last_frame_image).then((img) => img
134
+ ? {
135
+ kind: 'frame',
136
+ entry: {
137
+ type: 'image_url',
138
+ image_url: { url: `data:${img.mime};base64,${img.data}` },
139
+ frame_type: 'last_frame',
140
+ },
141
+ }
142
+ : null));
109
143
  }
144
+ const frameResults = await Promise.all(frameTasks);
145
+ const frameImages = frameResults
146
+ .filter((r) => r !== null)
147
+ .map((r) => r.entry);
110
148
  if (frameImages.length)
111
149
  body.frame_images = frameImages;
112
150
  if (args.reference_images?.length) {
113
- const refs = [];
114
- for (const src of args.reference_images) {
115
- const img = await prepareImageInput(src);
116
- if (img)
117
- refs.push({ image: { url: `data:${img.mime};base64,${img.data}` } });
118
- }
151
+ const refResults = await Promise.all(args.reference_images.map((src) => prepareImageInput(src)));
152
+ const refs = refResults
153
+ .filter((img) => img !== null)
154
+ .map((img) => ({
155
+ type: 'image_url',
156
+ image_url: { url: `data:${img.mime};base64,${img.data}` },
157
+ }));
119
158
  if (refs.length)
120
159
  body.input_references = refs;
121
160
  }
@@ -234,9 +273,10 @@ export async function handleGenerateVideo(request, apiClient, progress) {
234
273
  if (!args.prompt || !args.prompt.trim()) {
235
274
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
236
275
  }
237
- const model = args.model ||
238
- process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL ||
239
- FALLBACK_MODEL;
276
+ const model = args.model || process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL || FALLBACK_MODEL;
277
+ // Sora deprecation warning — OpenAI is removing the Videos API and all
278
+ // Sora 2 model aliases on September 24, 2026. Warn and suggest alternatives.
279
+ const deprecationWarning = checkSoraDeprecation(model);
240
280
  // Audit entry — video is the most expensive tool we have. Always log
241
281
  // model, resolution, duration, and a safe prompt preview so unintended
242
282
  // spend can be traced.
@@ -297,13 +337,16 @@ export async function handleGenerateVideo(request, apiClient, progress) {
297
337
  });
298
338
  }
299
339
  if (outcome.kind === 'timeout') {
340
+ const timeoutContent = [];
341
+ if (deprecationWarning) {
342
+ timeoutContent.push({ type: 'text', text: deprecationWarning });
343
+ }
344
+ timeoutContent.push({
345
+ type: 'text',
346
+ text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
347
+ });
300
348
  return {
301
- content: [
302
- {
303
- type: 'text',
304
- text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
305
- },
306
- ],
349
+ content: timeoutContent,
307
350
  isError: false,
308
351
  _meta: {
309
352
  server_version: SERVER_VERSION,
@@ -316,6 +359,11 @@ export async function handleGenerateVideo(request, apiClient, progress) {
316
359
  }
317
360
  try {
318
361
  const { content, _meta } = await finalizeCompletedJob(apiClient, outcome.status, safeSavePath);
362
+ // Prepend deprecation warning if applicable
363
+ if (deprecationWarning) {
364
+ content.unshift({ type: 'text', text: deprecationWarning });
365
+ _meta.deprecated_model = true;
366
+ }
319
367
  return { content, _meta };
320
368
  }
321
369
  catch (err) {
@@ -3,6 +3,7 @@ export declare const isBlockedIPv4: typeof _isBlockedIPv4;
3
3
  export declare const assertUrlSafeForFetch: typeof _assertUrlSafeForFetch;
4
4
  export declare function getMaxImageDimension(): number;
5
5
  export declare function getImageJpegQuality(): number;
6
+ export declare function mimeFromExtension(ext: string): string | null;
6
7
  export declare function getMimeType(filePath: string): string;
7
8
  export declare function fetchHttpImage(urlString: string): Promise<Buffer>;
8
9
  export declare function fetchImage(source: string): Promise<Buffer>;
@@ -1,6 +1,7 @@
1
1
  import path from 'path';
2
2
  import { promises as fs } from 'fs';
3
3
  import { readEnvInt, isBlockedIPv4 as _isBlockedIPv4, assertUrlSafeForFetch as _assertUrlSafeForFetch, fetchHttpResource, parseBase64DataUrl, } from './fetch-utils.js';
4
+ import { resolveSafeInputPath } from './path-safety.js';
4
5
  // Re-export for backward compatibility (tests import from image-utils)
5
6
  export const isBlockedIPv4 = _isBlockedIPv4;
6
7
  export const assertUrlSafeForFetch = _assertUrlSafeForFetch;
@@ -39,22 +40,29 @@ async function loadSharp() {
39
40
  sharpFn = fn ?? mod;
40
41
  }
41
42
  catch {
42
- console.error('sharp not available, images will be sent unprocessed');
43
+ // sharp is optional — images will be sent unprocessed (larger but functional)
44
+ const { logger } = await import('../logger.js');
45
+ logger.warn('sharp not available, images will be sent unprocessed');
43
46
  }
44
47
  }
45
48
  return sharpFn;
46
49
  }
50
+ const IMAGE_EXT_MIME = {
51
+ png: 'image/png',
52
+ jpg: 'image/jpeg',
53
+ jpeg: 'image/jpeg',
54
+ webp: 'image/webp',
55
+ gif: 'image/gif',
56
+ bmp: 'image/bmp',
57
+ };
58
+ export function mimeFromExtension(ext) {
59
+ const normalized = ext.toLowerCase().replace(/^\./, '');
60
+ if (!normalized)
61
+ return null;
62
+ return IMAGE_EXT_MIME[normalized] ?? null;
63
+ }
47
64
  export function getMimeType(filePath) {
48
- const ext = path.extname(filePath).toLowerCase();
49
- const map = {
50
- '.png': 'image/png',
51
- '.jpg': 'image/jpeg',
52
- '.jpeg': 'image/jpeg',
53
- '.webp': 'image/webp',
54
- '.gif': 'image/gif',
55
- '.bmp': 'image/bmp',
56
- };
57
- return map[ext] || 'image/jpeg';
65
+ return mimeFromExtension(path.extname(filePath)) ?? 'image/jpeg';
58
66
  }
59
67
  export async function fetchHttpImage(urlString) {
60
68
  const { buffer } = await fetchHttpResource(urlString, {
@@ -77,7 +85,8 @@ export async function fetchImage(source) {
77
85
  if (source.startsWith('http://') || source.startsWith('https://')) {
78
86
  return fetchHttpImage(source);
79
87
  }
80
- return fs.readFile(source);
88
+ const safe = await resolveSafeInputPath(source);
89
+ return fs.readFile(safe);
81
90
  }
82
91
  /**
83
92
  * Sniff image MIME type from magic bytes. Used to label the output of a
@@ -134,13 +143,14 @@ export async function optimizeImage(buffer) {
134
143
  const maxDim = getMaxImageDimension();
135
144
  const quality = getImageJpegQuality();
136
145
  try {
137
- const meta = await sharp(buffer).metadata();
138
- let pipeline = sharp(buffer);
146
+ const pipeline = sharp(buffer);
147
+ const meta = await pipeline.metadata();
148
+ let resized = pipeline;
139
149
  if (meta.width && meta.height && Math.max(meta.width, meta.height) > maxDim) {
140
150
  const opts = meta.width > meta.height ? { width: maxDim } : { height: maxDim };
141
- pipeline = pipeline.resize(opts);
151
+ resized = pipeline.resize(opts);
142
152
  }
143
- const out = await pipeline.jpeg({ quality }).toBuffer();
153
+ const out = await resized.jpeg({ quality }).toBuffer();
144
154
  return { base64: out.toString('base64'), mime: 'image/jpeg' };
145
155
  }
146
156
  catch {
@@ -110,7 +110,9 @@ export function classifyUpstreamError(err, contextMessage) {
110
110
  }
111
111
  // Model lookup failures.
112
112
  if (lower.includes('model') &&
113
- (lower.includes('does not exist') || lower.includes('not found') || lower.includes('invalid model'))) {
113
+ (lower.includes('does not exist') ||
114
+ lower.includes('not found') ||
115
+ lower.includes('invalid model'))) {
114
116
  return toolError(ErrorCode.MODEL_NOT_FOUND, fullMsg, { status }, {
115
117
  suggestions: [
116
118
  'Use search_models to discover valid model ids',
@@ -119,7 +121,9 @@ export function classifyUpstreamError(err, contextMessage) {
119
121
  });
120
122
  }
121
123
  // Content policy / moderation — surface as UPSTREAM_REFUSED so callers can distinguish from 5xx.
122
- if (lower.includes('content policy') || lower.includes('moderation') || lower.includes('refused')) {
124
+ if (lower.includes('content policy') ||
125
+ lower.includes('moderation') ||
126
+ lower.includes('refused')) {
123
127
  return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'policy' }, {
124
128
  suggestions: ['Rephrase the prompt', 'Try a different provider via provider.order'],
125
129
  });