@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +367 -283
  2. package/dist/index.js +1 -1
  3. package/dist/model-cache.d.ts +22 -12
  4. package/dist/model-cache.js +58 -21
  5. package/dist/tool-descriptions.d.ts +19 -0
  6. package/dist/tool-descriptions.js +423 -0
  7. package/dist/tool-handlers/analyze-audio.js +5 -1
  8. package/dist/tool-handlers/analyze-image.js +6 -5
  9. package/dist/tool-handlers/analyze-video.js +6 -5
  10. package/dist/tool-handlers/audio-utils.js +4 -2
  11. package/dist/tool-handlers/chat-completion.js +1 -1
  12. package/dist/tool-handlers/fetch-utils.js +16 -2
  13. package/dist/tool-handlers/generate-audio.js +2 -4
  14. package/dist/tool-handlers/generate-image-input.d.ts +3 -0
  15. package/dist/tool-handlers/generate-image-input.js +38 -0
  16. package/dist/tool-handlers/generate-image.d.ts +13 -51
  17. package/dist/tool-handlers/generate-image.js +32 -119
  18. package/dist/tool-handlers/generate-video.js +28 -24
  19. package/dist/tool-handlers/image-utils.d.ts +1 -0
  20. package/dist/tool-handlers/image-utils.js +26 -16
  21. package/dist/tool-handlers/openrouter-errors.js +6 -2
  22. package/dist/tool-handlers/provider-routing.js +7 -2
  23. package/dist/tool-handlers/rerank.js +2 -5
  24. package/dist/tool-handlers/search-models.d.ts +2 -2
  25. package/dist/tool-handlers/search-models.js +2 -6
  26. package/dist/tool-handlers/structured-output.d.ts +8 -0
  27. package/dist/tool-handlers/structured-output.js +11 -0
  28. package/dist/tool-handlers/video-utils.js +6 -9
  29. package/dist/tool-handlers.js +25 -123
  30. package/dist/version.d.ts +1 -1
  31. package/dist/version.js +1 -1
  32. package/package.json +26 -14
@@ -1,10 +1,11 @@
1
1
  import { prepareVideoData } from './video-utils.js';
2
+ import { UnsafeOutputPathError } from './path-safety.js';
2
3
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
3
4
  import { SERVER_VERSION } from '../version.js';
4
5
  import { logger } from '../logger.js';
5
6
  import { classifyUpstreamError } from './openrouter-errors.js';
6
7
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
7
- import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
8
+ import { buildCacheHeaders, extractCacheMeta } from './cache.js';
8
9
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
9
10
  /**
10
11
  * Default model — `google/gemini-2.5-flash` has the widest video-input
@@ -18,15 +19,15 @@ export async function handleAnalyzeVideo(request, openai, defaultModel) {
18
19
  if (!video_path) {
19
20
  return toolError(ErrorCode.INVALID_INPUT, 'video_path is required.');
20
21
  }
21
- const pickedModel = model ||
22
- process.env.OPENROUTER_DEFAULT_VIDEO_MODEL ||
23
- defaultModel ||
24
- FALLBACK_DEFAULT_MODEL;
22
+ const pickedModel = model || process.env.OPENROUTER_DEFAULT_VIDEO_MODEL || defaultModel || FALLBACK_DEFAULT_MODEL;
25
23
  let videoData;
26
24
  try {
27
25
  videoData = await prepareVideoData(video_path);
28
26
  }
29
27
  catch (err) {
28
+ if (err instanceof UnsafeOutputPathError) {
29
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
30
+ }
30
31
  const msg = err instanceof Error ? err.message : String(err);
31
32
  if (msg.includes('Blocked host')) {
32
33
  return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
@@ -5,6 +5,7 @@
5
5
  import path from 'path';
6
6
  import { promises as fs } from 'fs';
7
7
  import { readEnvInt, fetchHttpResource, parseBase64DataUrl } from './fetch-utils.js';
8
+ import { resolveSafeInputPath } from './path-safety.js';
8
9
  // Re-export for tests
9
10
  export { isBlockedIPv4, assertUrlSafeForFetch } from './fetch-utils.js';
10
11
  const DEFAULT_FETCH_TIMEOUT_MS = 30_000;
@@ -117,10 +118,11 @@ export async function prepareAudioData(source) {
117
118
  return { data: buffer.toString('base64'), format };
118
119
  }
119
120
  // --- local file ---
120
- const format = getAudioFormat(source);
121
+ const safe = await resolveSafeInputPath(source);
122
+ const format = getAudioFormat(safe);
121
123
  if (!format) {
122
124
  throw new Error(`Unsupported audio format for file: ${source}. Supported: ${SUPPORTED_AUDIO_FORMATS.join(', ')}`);
123
125
  }
124
- const buffer = await fs.readFile(source);
126
+ const buffer = await fs.readFile(safe);
125
127
  return { data: buffer.toString('base64'), format };
126
128
  }
@@ -3,7 +3,7 @@ import { SERVER_VERSION } from '../version.js';
3
3
  import { classifyUpstreamError } from './openrouter-errors.js';
4
4
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
5
5
  import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
6
- import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
6
+ import { buildCacheHeaders, extractCacheMeta } from './cache.js';
7
7
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
8
8
  const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
9
9
  function readIncludeReasoningDefault() {
@@ -151,11 +151,25 @@ export function isBlockedIPv6(ip) {
151
151
  return false;
152
152
  const [g0, g1, g2, g3, g4, g5, g6, g7] = groups;
153
153
  // :: (unspecified)
154
- if (g0 === 0 && g1 === 0 && g2 === 0 && g3 === 0 && g4 === 0 && g5 === 0 && g6 === 0 && g7 === 0) {
154
+ if (g0 === 0 &&
155
+ g1 === 0 &&
156
+ g2 === 0 &&
157
+ g3 === 0 &&
158
+ g4 === 0 &&
159
+ g5 === 0 &&
160
+ g6 === 0 &&
161
+ g7 === 0) {
155
162
  return true;
156
163
  }
157
164
  // ::1 (loopback)
158
- if (g0 === 0 && g1 === 0 && g2 === 0 && g3 === 0 && g4 === 0 && g5 === 0 && g6 === 0 && g7 === 1) {
165
+ if (g0 === 0 &&
166
+ g1 === 0 &&
167
+ g2 === 0 &&
168
+ g3 === 0 &&
169
+ g4 === 0 &&
170
+ g5 === 0 &&
171
+ g6 === 0 &&
172
+ g7 === 1) {
159
173
  return true;
160
174
  }
161
175
  // ::ffff:0:0/96 — IPv4-mapped. Re-check the embedded IPv4.
@@ -95,9 +95,7 @@ export async function handleGenerateAudio(request, openai) {
95
95
  logger.audit('generate_audio.start', {
96
96
  model: model || DEFAULT_MODEL,
97
97
  voice: voice?.trim() || DEFAULT_VOICE,
98
- format: VALID_FORMATS.includes(format ?? '')
99
- ? format
100
- : DEFAULT_FORMAT,
98
+ format: VALID_FORMATS.includes(format ?? '') ? format : DEFAULT_FORMAT,
101
99
  prompt_preview: prompt.slice(0, 80),
102
100
  save_path: save_path ? 'provided' : 'none',
103
101
  });
@@ -156,7 +154,7 @@ export async function handleGenerateAudio(request, openai) {
156
154
  const detected = detectAudioFormat(audioBuffer);
157
155
  // Always wrap raw PCM in WAV so it's playable
158
156
  if (detected.ext === 'pcm') {
159
- audioBuffer = wrapPcmInWav(audioBuffer);
157
+ audioBuffer = Buffer.from(wrapPcmInWav(audioBuffer));
160
158
  detected.ext = 'wav';
161
159
  detected.mimeType = 'audio/wav';
162
160
  }
@@ -0,0 +1,3 @@
1
+ import OpenAI from 'openai';
2
+ export declare function resolveInputImage(ref: string): Promise<string>;
3
+ export declare function buildUserContent(prompt: string, inputImages?: string[]): Promise<string | OpenAI.Chat.Completions.ChatCompletionContentPart[]>;
@@ -0,0 +1,38 @@
1
+ import { promises as fs } from 'fs';
2
+ import path from 'node:path';
3
+ import { resolveSafeInputPath } from './path-safety.js';
4
+ import { mimeFromExtension } from './image-utils.js';
5
+ export async function resolveInputImage(ref) {
6
+ const trimmed = ref.trim();
7
+ if (!trimmed)
8
+ throw new Error('empty input_images entry');
9
+ if (trimmed.startsWith('data:'))
10
+ return trimmed;
11
+ if (/^https?:\/\//i.test(trimmed))
12
+ return trimmed;
13
+ const abs = await resolveSafeInputPath(trimmed);
14
+ const buf = await fs.readFile(abs);
15
+ const mime = mimeFromExtension(path.extname(abs)) || 'image/png';
16
+ return `data:${mime};base64,${buf.toString('base64')}`;
17
+ }
18
+ export async function buildUserContent(prompt, inputImages) {
19
+ if (!inputImages?.length) {
20
+ return `Generate an image: ${prompt}`;
21
+ }
22
+ const parts = [
23
+ {
24
+ type: 'text',
25
+ text: `Generate an image based on this prompt, using the following reference image(s) ` +
26
+ `for visual consistency. Match the appearance, identity, and style of the references ` +
27
+ `closely; do not alter them.\n\nPrompt: ${prompt}`,
28
+ },
29
+ ];
30
+ const urls = await Promise.all(inputImages.map((ref) => resolveInputImage(ref)));
31
+ for (const url of urls) {
32
+ parts.push({
33
+ type: 'image_url',
34
+ image_url: { url, detail: 'high' },
35
+ });
36
+ }
37
+ return parts;
38
+ }
@@ -3,48 +3,10 @@ export interface GenerateImageToolRequest {
3
3
  prompt: string;
4
4
  model?: string;
5
5
  save_path?: string;
6
- /**
7
- * Output aspect ratio. Passed through as `image_config.aspect_ratio`.
8
- * Supported by OpenRouter image models (e.g. `1:1`, `16:9`, `9:16`,
9
- * `4:3`, `3:4`, `21:9`). Model-dependent. Unsupported values fall back
10
- * to the model's default. See
11
- * https://openrouter.ai/docs/guides/overview/multimodal/image-generation
12
- */
13
6
  aspect_ratio?: string;
14
- /**
15
- * Output image resolution bucket. Passed through as
16
- * `image_config.image_size`. Typical values: `0.5K`, `1K` (default),
17
- * `2K`, `4K`. Model-dependent.
18
- */
19
7
  image_size?: string;
20
- /**
21
- * Upper bound on the completion budget. Without this OpenRouter
22
- * reserves the model's full context window (~29k for Gemini
23
- * image models), which can trigger a 402 on low-credit accounts even
24
- * though the actual generation uses far fewer tokens. 4096 is plenty
25
- * for the image payload + any caption.
26
- */
27
8
  max_tokens?: number;
28
- /**
29
- * Optional reference images. Each entry is one of:
30
- * - a `data:image/...;base64,...` URL,
31
- * - an `http(s)://` URL (OpenRouter fetches it),
32
- * - a local file path (sandboxed to `OPENROUTER_INPUT_DIR` /
33
- * `OPENROUTER_OUTPUT_DIR` / cwd, read + base64-encoded by the server).
34
- *
35
- * When provided, the user message becomes multimodal: a text prompt
36
- * plus one `image_url` block per reference, in array order. Enables
37
- * character / style consistency and image-to-image refinement on
38
- * chat-image models that accept image inputs (Gemini Nano Banana,
39
- * `openai/gpt-5.4-image-2`).
40
- */
41
9
  input_images?: string[];
42
- /**
43
- * Override the default `modalities: ["image","text"]` sent to
44
- * OpenRouter. Most callers should leave this unset. Provide e.g.
45
- * `["text"]` to suppress image output for inspection / captioning,
46
- * or other shapes for future model variants.
47
- */
48
10
  modalities?: string[];
49
11
  }
50
12
  export declare function handleGenerateImage(request: {
@@ -64,11 +26,16 @@ export declare function handleGenerateImage(request: {
64
26
  text?: undefined;
65
27
  })[];
66
28
  _meta: {
67
- usage?: {
29
+ usage: {
68
30
  prompt_tokens: number;
69
31
  completion_tokens: number;
70
32
  total_tokens: number;
71
- } | undefined;
33
+ };
34
+ server_version: string;
35
+ save_path: string;
36
+ mime: string;
37
+ } | {
38
+ usage?: undefined;
72
39
  server_version: string;
73
40
  save_path: string;
74
41
  mime: string;
@@ -80,21 +47,16 @@ export declare function handleGenerateImage(request: {
80
47
  data: string;
81
48
  }[];
82
49
  _meta: {
83
- usage?: {
50
+ usage: {
84
51
  prompt_tokens: number;
85
52
  completion_tokens: number;
86
53
  total_tokens: number;
87
- } | undefined;
54
+ };
55
+ server_version: string;
56
+ mime: string;
57
+ } | {
58
+ usage?: undefined;
88
59
  server_version: string;
89
60
  mime: string;
90
61
  };
91
62
  }>;
92
- /**
93
- * Resolve a caller-supplied input image into a URL the chat-completions
94
- * API accepts. Local file paths are sandboxed via
95
- * `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
96
- * `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
97
- */
98
- export declare function resolveInputImage(ref: string): Promise<string>;
99
- export declare function mimeFromExt(ext: string): string | null;
100
- export declare function buildUserContent(prompt: string, inputImages?: string[]): Promise<string | OpenAI.Chat.Completions.ChatCompletionContentPart[]>;
@@ -1,15 +1,12 @@
1
1
  import { promises as fs } from 'fs';
2
- import path from 'node:path';
3
- import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
2
+ import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
4
3
  import { parseBase64DataUrl } from './fetch-utils.js';
4
+ import { buildUserContent } from './generate-image-input.js';
5
5
  import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
6
6
  import { SERVER_VERSION } from '../version.js';
7
7
  import { logger } from '../logger.js';
8
8
  import { classifyUpstreamError } from './openrouter-errors.js';
9
9
  const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
10
- // OpenRouter-documented aspect ratios (standard + extended). Extended are
11
- // only honored by models that support them (e.g. gemini-3.1-flash-image),
12
- // others fall back to the model's default.
13
10
  const VALID_ASPECT_RATIOS = new Set([
14
11
  '1:1',
15
12
  '2:3',
@@ -32,9 +29,6 @@ export async function handleGenerateImage(request, openai) {
32
29
  if (!prompt?.trim()) {
33
30
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
34
31
  }
35
- // Audit entry. Bypasses the normal log level so operators always see a
36
- // record of cost-incurring operations. Prompt preview is hard-capped
37
- // at 80 chars to avoid PII spillage in log aggregators.
38
32
  logger.audit('generate_image.start', {
39
33
  model: model || DEFAULT_MODEL,
40
34
  prompt_preview: prompt.slice(0, 80),
@@ -43,15 +37,12 @@ export async function handleGenerateImage(request, openai) {
43
37
  save_path: save_path ? 'provided' : 'none',
44
38
  input_images_count: input_images?.length ?? 0,
45
39
  });
46
- // Validate optional shape fields early so callers get a clear error
47
- // instead of a cryptic upstream 400.
48
40
  if (aspect_ratio !== undefined && !VALID_ASPECT_RATIOS.has(aspect_ratio)) {
49
- return toolError(ErrorCode.INVALID_INPUT, `aspect_ratio '${aspect_ratio}' is not supported. Valid values: ${[...VALID_ASPECT_RATIOS].join(', ')}.`);
41
+ return invalidEnumError('aspect_ratio', aspect_ratio, VALID_ASPECT_RATIOS);
50
42
  }
51
43
  if (image_size !== undefined && !VALID_IMAGE_SIZES.has(image_size)) {
52
- return toolError(ErrorCode.INVALID_INPUT, `image_size '${image_size}' is not supported. Valid values: ${[...VALID_IMAGE_SIZES].join(', ')}.`);
44
+ return invalidEnumError('image_size', image_size, VALID_IMAGE_SIZES);
53
45
  }
54
- // Fail-fast on unsafe paths BEFORE spending tokens.
55
46
  let safePathResolved = null;
56
47
  if (save_path) {
57
48
  try {
@@ -64,9 +55,6 @@ export async function handleGenerateImage(request, openai) {
64
55
  return toolErrorFrom(ErrorCode.INTERNAL, err);
65
56
  }
66
57
  }
67
- // Build the user message. With no `input_images`, this is the original
68
- // string content; with refs, it becomes a multimodal
69
- // ChatCompletionContentPart[] (text preamble + one image_url per ref).
70
58
  let content;
71
59
  try {
72
60
  content = await buildUserContent(prompt, input_images);
@@ -77,14 +65,6 @@ export async function handleGenerateImage(request, openai) {
77
65
  }
78
66
  return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_images');
79
67
  }
80
- // Assemble the request body. OpenRouter's image-generation guide
81
- // requires:
82
- // - `modalities: ["image", "text"]` so multimodal models (like
83
- // Gemini) know to emit an image, not just text. Caller can
84
- // override via the `modalities` field.
85
- // - `image_config.{aspect_ratio, image_size}` for shape control.
86
- // The OpenAI SDK doesn't type these fields, but passes unknown members
87
- // through to the server, so we attach them via a typed cast.
88
68
  const imageConfig = {};
89
69
  if (aspect_ratio)
90
70
  imageConfig.aspect_ratio = aspect_ratio;
@@ -93,7 +73,7 @@ export async function handleGenerateImage(request, openai) {
93
73
  const body = {
94
74
  model: model || DEFAULT_MODEL,
95
75
  messages: [{ role: 'user', content }],
96
- modalities: modalities && modalities.length ? modalities : ['image', 'text'],
76
+ modalities: modalities?.length ? modalities : ['image', 'text'],
97
77
  };
98
78
  if (Object.keys(imageConfig).length > 0)
99
79
  body.image_config = imageConfig;
@@ -101,10 +81,6 @@ export async function handleGenerateImage(request, openai) {
101
81
  body.max_tokens = max_tokens;
102
82
  let completion;
103
83
  try {
104
- // OpenRouter-specific `image_config` isn't in the OpenAI SDK's typings,
105
- // but the SDK passes unknown fields straight through to the server.
106
- // We never pass `stream: true`, so the response is always
107
- // ChatCompletion.
108
84
  completion = (await openai.chat.completions.create(body));
109
85
  }
110
86
  catch (err) {
@@ -116,8 +92,6 @@ export async function handleGenerateImage(request, openai) {
116
92
  }
117
93
  const base64 = extractBase64(message);
118
94
  if (!base64) {
119
- // Model talked but did not emit an image. Surface this as a distinct
120
- // condition so callers don't treat chatter as a successful image.
121
95
  const messageContent = message.content;
122
96
  const text = typeof messageContent === 'string' ? messageContent : JSON.stringify(messageContent);
123
97
  return toolError(ErrorCode.UPSTREAM_REFUSED, `Model returned no image. Text response: ${text.slice(0, 300)}`, {
@@ -127,113 +101,56 @@ export async function handleGenerateImage(request, openai) {
127
101
  }
128
102
  if (safePathResolved) {
129
103
  try {
130
- await fs.writeFile(safePathResolved, Buffer.from(base64.data, 'base64'));
104
+ await fs.writeFile(safePathResolved, base64.data, { encoding: 'base64' });
131
105
  }
132
106
  catch (err) {
133
107
  return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
134
108
  }
135
- const usage = completion.usage;
109
+ }
110
+ return buildImageSuccessResult(base64, completion.usage, safePathResolved ?? undefined);
111
+ }
112
+ function invalidEnumError(field, value, allowed) {
113
+ return toolError(ErrorCode.INVALID_INPUT, `${field} '${value}' is not supported. Valid values: ${[...allowed].join(', ')}.`);
114
+ }
115
+ function buildImageSuccessResult(base64, usage, savePath) {
116
+ const usageMeta = usage
117
+ ? {
118
+ usage: {
119
+ prompt_tokens: usage.prompt_tokens,
120
+ completion_tokens: usage.completion_tokens,
121
+ total_tokens: usage.total_tokens,
122
+ },
123
+ }
124
+ : {};
125
+ if (savePath) {
136
126
  return {
137
127
  content: [
138
- { type: 'text', text: `Image saved to: ${safePathResolved}` },
128
+ { type: 'text', text: `Image saved to: ${savePath}` },
139
129
  { type: 'image', mimeType: base64.mime, data: base64.data },
140
130
  ],
141
131
  _meta: {
142
132
  server_version: SERVER_VERSION,
143
- save_path: safePathResolved,
133
+ save_path: savePath,
144
134
  mime: base64.mime,
145
- ...(usage
146
- ? {
147
- usage: {
148
- prompt_tokens: usage.prompt_tokens,
149
- completion_tokens: usage.completion_tokens,
150
- total_tokens: usage.total_tokens,
151
- },
152
- }
153
- : {}),
135
+ ...usageMeta,
154
136
  },
155
137
  };
156
138
  }
157
- const usage = completion.usage;
158
139
  return {
159
140
  content: [{ type: 'image', mimeType: base64.mime, data: base64.data }],
160
141
  _meta: {
161
142
  server_version: SERVER_VERSION,
162
143
  mime: base64.mime,
163
- ...(usage
164
- ? {
165
- usage: {
166
- prompt_tokens: usage.prompt_tokens,
167
- completion_tokens: usage.completion_tokens,
168
- total_tokens: usage.total_tokens,
169
- },
170
- }
171
- : {}),
144
+ ...usageMeta,
172
145
  },
173
146
  };
174
147
  }
175
- /**
176
- * Resolve a caller-supplied input image into a URL the chat-completions
177
- * API accepts. Local file paths are sandboxed via
178
- * `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
179
- * `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
180
- */
181
- export async function resolveInputImage(ref) {
182
- const trimmed = ref.trim();
183
- if (!trimmed)
184
- throw new Error('empty input_images entry');
185
- if (trimmed.startsWith('data:'))
186
- return trimmed;
187
- if (/^https?:\/\//i.test(trimmed))
188
- return trimmed;
189
- const abs = await resolveSafeInputPath(trimmed);
190
- const buf = await fs.readFile(abs);
191
- const mime = mimeFromExt(path.extname(abs)) || 'image/png';
192
- return `data:${mime};base64,${buf.toString('base64')}`;
193
- }
194
- export function mimeFromExt(ext) {
195
- const e = ext.toLowerCase().replace(/^\./, '');
196
- switch (e) {
197
- case 'png':
198
- return 'image/png';
199
- case 'jpg':
200
- case 'jpeg':
201
- return 'image/jpeg';
202
- case 'webp':
203
- return 'image/webp';
204
- case 'gif':
205
- return 'image/gif';
206
- default:
207
- return null;
208
- }
209
- }
210
- export async function buildUserContent(prompt, inputImages) {
211
- if (!inputImages?.length) {
212
- return `Generate an image: ${prompt}`;
213
- }
214
- const parts = [
215
- {
216
- type: 'text',
217
- text: `Generate an image based on this prompt, using the following reference image(s) ` +
218
- `for visual consistency. Match the appearance, identity, and style of the references ` +
219
- `closely; do not alter them.\n\nPrompt: ${prompt}`,
220
- },
221
- ];
222
- for (const ref of inputImages) {
223
- const url = await resolveInputImage(ref);
224
- parts.push({
225
- type: 'image_url',
226
- image_url: { url, detail: 'high' },
227
- });
228
- }
229
- return parts;
230
- }
231
148
  function extractBase64(message) {
232
149
  const images = message.images;
233
150
  if (Array.isArray(images) && images.length) {
234
151
  for (const img of images) {
235
152
  const imageUrl = img.image_url;
236
- const result = parseDataUrl(imageUrl?.url || img.url);
153
+ const result = dataUrlToBase64(imageUrl?.url || img.url);
237
154
  if (result)
238
155
  return result;
239
156
  }
@@ -243,7 +160,7 @@ function extractBase64(message) {
243
160
  const iu = part.image_url;
244
161
  const url = iu?.url || part.url;
245
162
  if (url) {
246
- const r = parseDataUrl(url);
163
+ const r = dataUrlToBase64(url);
247
164
  if (r)
248
165
  return r;
249
166
  }
@@ -257,24 +174,20 @@ function extractBase64(message) {
257
174
  }
258
175
  }
259
176
  if (typeof message.content === 'string') {
260
- // Scan the string for an embedded data URL. We deliberately don't use
261
- // a single regex here because data URLs may carry MIME parameters
262
- // (e.g. `data:image/png;charset=binary;base64,...`) which trips the
263
- // naive `data:([^;]+);base64,(.+)` form.
177
+ // Data URLs may carry MIME parameters (e.g. charset=binary) that break naive regexes.
264
178
  const start = message.content.indexOf('data:image/');
265
179
  if (start >= 0) {
266
- // Find the end of the data URL: a whitespace or closing quote/paren.
267
180
  const tail = message.content.slice(start);
268
181
  const end = tail.search(/[\s)"']/);
269
182
  const url = end === -1 ? tail : tail.slice(0, end);
270
- const parsed = parseDataUrl(url);
183
+ const parsed = dataUrlToBase64(url);
271
184
  if (parsed)
272
185
  return parsed;
273
186
  }
274
187
  }
275
188
  return null;
276
189
  }
277
- function parseDataUrl(url) {
190
+ function dataUrlToBase64(url) {
278
191
  if (!url?.startsWith('data:'))
279
192
  return null;
280
193
  const parsed = parseBase64DataUrl(url);
@@ -88,34 +88,40 @@ function buildRequestBody(args, model) {
88
88
  return body;
89
89
  }
90
90
  async function attachFrameImages(args, body) {
91
- const frameImages = [];
91
+ const frameTasks = [];
92
92
  if (args.first_frame_image) {
93
- const img = await prepareImageInput(args.first_frame_image);
94
- if (img) {
95
- frameImages.push({
96
- frame_type: 'first_frame',
97
- image: { url: `data:${img.mime};base64,${img.data}` },
98
- });
99
- }
93
+ frameTasks.push(prepareImageInput(args.first_frame_image).then((img) => img
94
+ ? {
95
+ kind: 'frame',
96
+ entry: {
97
+ frame_type: 'first_frame',
98
+ image: { url: `data:${img.mime};base64,${img.data}` },
99
+ },
100
+ }
101
+ : null));
100
102
  }
101
103
  if (args.last_frame_image) {
102
- const img = await prepareImageInput(args.last_frame_image);
103
- if (img) {
104
- frameImages.push({
105
- frame_type: 'last_frame',
106
- image: { url: `data:${img.mime};base64,${img.data}` },
107
- });
108
- }
104
+ frameTasks.push(prepareImageInput(args.last_frame_image).then((img) => img
105
+ ? {
106
+ kind: 'frame',
107
+ entry: {
108
+ frame_type: 'last_frame',
109
+ image: { url: `data:${img.mime};base64,${img.data}` },
110
+ },
111
+ }
112
+ : null));
109
113
  }
114
+ const frameResults = await Promise.all(frameTasks);
115
+ const frameImages = frameResults
116
+ .filter((r) => r !== null)
117
+ .map((r) => r.entry);
110
118
  if (frameImages.length)
111
119
  body.frame_images = frameImages;
112
120
  if (args.reference_images?.length) {
113
- const refs = [];
114
- for (const src of args.reference_images) {
115
- const img = await prepareImageInput(src);
116
- if (img)
117
- refs.push({ image: { url: `data:${img.mime};base64,${img.data}` } });
118
- }
121
+ const refResults = await Promise.all(args.reference_images.map((src) => prepareImageInput(src)));
122
+ const refs = refResults
123
+ .filter((img) => img !== null)
124
+ .map((img) => ({ image: { url: `data:${img.mime};base64,${img.data}` } }));
119
125
  if (refs.length)
120
126
  body.input_references = refs;
121
127
  }
@@ -234,9 +240,7 @@ export async function handleGenerateVideo(request, apiClient, progress) {
234
240
  if (!args.prompt || !args.prompt.trim()) {
235
241
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
236
242
  }
237
- const model = args.model ||
238
- process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL ||
239
- FALLBACK_MODEL;
243
+ const model = args.model || process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL || FALLBACK_MODEL;
240
244
  // Audit entry — video is the most expensive tool we have. Always log
241
245
  // model, resolution, duration, and a safe prompt preview so unintended
242
246
  // spend can be traced.
@@ -3,6 +3,7 @@ export declare const isBlockedIPv4: typeof _isBlockedIPv4;
3
3
  export declare const assertUrlSafeForFetch: typeof _assertUrlSafeForFetch;
4
4
  export declare function getMaxImageDimension(): number;
5
5
  export declare function getImageJpegQuality(): number;
6
+ export declare function mimeFromExtension(ext: string): string | null;
6
7
  export declare function getMimeType(filePath: string): string;
7
8
  export declare function fetchHttpImage(urlString: string): Promise<Buffer>;
8
9
  export declare function fetchImage(source: string): Promise<Buffer>;