@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +367 -283
- package/dist/index.js +1 -1
- package/dist/model-cache.d.ts +22 -12
- package/dist/model-cache.js +58 -21
- package/dist/openrouter-api.d.ts +45 -0
- package/dist/openrouter-api.js +50 -0
- package/dist/tool-descriptions.d.ts +19 -0
- package/dist/tool-descriptions.js +573 -0
- package/dist/tool-handlers/analyze-audio.js +5 -1
- package/dist/tool-handlers/analyze-image.js +6 -5
- package/dist/tool-handlers/analyze-video.js +6 -5
- package/dist/tool-handlers/async-chat.d.ts +51 -0
- package/dist/tool-handlers/async-chat.js +216 -0
- package/dist/tool-handlers/audio-utils.js +4 -2
- package/dist/tool-handlers/chat-completion.js +1 -1
- package/dist/tool-handlers/fetch-utils.js +16 -2
- package/dist/tool-handlers/generate-audio.js +2 -4
- package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
- package/dist/tool-handlers/generate-image-dedicated.js +176 -0
- package/dist/tool-handlers/generate-image-input.d.ts +3 -0
- package/dist/tool-handlers/generate-image-input.js +38 -0
- package/dist/tool-handlers/generate-image.d.ts +13 -51
- package/dist/tool-handlers/generate-image.js +32 -119
- package/dist/tool-handlers/generate-video.d.ts +2 -2
- package/dist/tool-handlers/generate-video.js +78 -30
- package/dist/tool-handlers/image-utils.d.ts +1 -0
- package/dist/tool-handlers/image-utils.js +26 -16
- package/dist/tool-handlers/openrouter-errors.js +6 -2
- package/dist/tool-handlers/path-safety.js +32 -5
- package/dist/tool-handlers/provider-routing.js +7 -2
- package/dist/tool-handlers/rerank.js +2 -5
- package/dist/tool-handlers/search-models.d.ts +2 -2
- package/dist/tool-handlers/search-models.js +2 -6
- package/dist/tool-handlers/speech-to-text.d.ts +20 -0
- package/dist/tool-handlers/speech-to-text.js +140 -0
- package/dist/tool-handlers/structured-output.d.ts +8 -0
- package/dist/tool-handlers/structured-output.js +11 -0
- package/dist/tool-handlers/text-to-speech.d.ts +29 -0
- package/dist/tool-handlers/text-to-speech.js +105 -0
- package/dist/tool-handlers/video-utils.js +6 -9
- package/dist/tool-handlers.js +253 -125
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +27 -15
|
@@ -3,48 +3,10 @@ export interface GenerateImageToolRequest {
|
|
|
3
3
|
prompt: string;
|
|
4
4
|
model?: string;
|
|
5
5
|
save_path?: string;
|
|
6
|
-
/**
|
|
7
|
-
* Output aspect ratio. Passed through as `image_config.aspect_ratio`.
|
|
8
|
-
* Supported by OpenRouter image models (e.g. `1:1`, `16:9`, `9:16`,
|
|
9
|
-
* `4:3`, `3:4`, `21:9`). Model-dependent. Unsupported values fall back
|
|
10
|
-
* to the model's default. See
|
|
11
|
-
* https://openrouter.ai/docs/guides/overview/multimodal/image-generation
|
|
12
|
-
*/
|
|
13
6
|
aspect_ratio?: string;
|
|
14
|
-
/**
|
|
15
|
-
* Output image resolution bucket. Passed through as
|
|
16
|
-
* `image_config.image_size`. Typical values: `0.5K`, `1K` (default),
|
|
17
|
-
* `2K`, `4K`. Model-dependent.
|
|
18
|
-
*/
|
|
19
7
|
image_size?: string;
|
|
20
|
-
/**
|
|
21
|
-
* Upper bound on the completion budget. Without this OpenRouter
|
|
22
|
-
* reserves the model's full context window (~29k for Gemini
|
|
23
|
-
* image models), which can trigger a 402 on low-credit accounts even
|
|
24
|
-
* though the actual generation uses far fewer tokens. 4096 is plenty
|
|
25
|
-
* for the image payload + any caption.
|
|
26
|
-
*/
|
|
27
8
|
max_tokens?: number;
|
|
28
|
-
/**
|
|
29
|
-
* Optional reference images. Each entry is one of:
|
|
30
|
-
* - a `data:image/...;base64,...` URL,
|
|
31
|
-
* - an `http(s)://` URL (OpenRouter fetches it),
|
|
32
|
-
* - a local file path (sandboxed to `OPENROUTER_INPUT_DIR` /
|
|
33
|
-
* `OPENROUTER_OUTPUT_DIR` / cwd, read + base64-encoded by the server).
|
|
34
|
-
*
|
|
35
|
-
* When provided, the user message becomes multimodal: a text prompt
|
|
36
|
-
* plus one `image_url` block per reference, in array order. Enables
|
|
37
|
-
* character / style consistency and image-to-image refinement on
|
|
38
|
-
* chat-image models that accept image inputs (Gemini Nano Banana,
|
|
39
|
-
* `openai/gpt-5.4-image-2`).
|
|
40
|
-
*/
|
|
41
9
|
input_images?: string[];
|
|
42
|
-
/**
|
|
43
|
-
* Override the default `modalities: ["image","text"]` sent to
|
|
44
|
-
* OpenRouter. Most callers should leave this unset. Provide e.g.
|
|
45
|
-
* `["text"]` to suppress image output for inspection / captioning,
|
|
46
|
-
* or other shapes for future model variants.
|
|
47
|
-
*/
|
|
48
10
|
modalities?: string[];
|
|
49
11
|
}
|
|
50
12
|
export declare function handleGenerateImage(request: {
|
|
@@ -64,11 +26,16 @@ export declare function handleGenerateImage(request: {
|
|
|
64
26
|
text?: undefined;
|
|
65
27
|
})[];
|
|
66
28
|
_meta: {
|
|
67
|
-
usage
|
|
29
|
+
usage: {
|
|
68
30
|
prompt_tokens: number;
|
|
69
31
|
completion_tokens: number;
|
|
70
32
|
total_tokens: number;
|
|
71
|
-
}
|
|
33
|
+
};
|
|
34
|
+
server_version: string;
|
|
35
|
+
save_path: string;
|
|
36
|
+
mime: string;
|
|
37
|
+
} | {
|
|
38
|
+
usage?: undefined;
|
|
72
39
|
server_version: string;
|
|
73
40
|
save_path: string;
|
|
74
41
|
mime: string;
|
|
@@ -80,21 +47,16 @@ export declare function handleGenerateImage(request: {
|
|
|
80
47
|
data: string;
|
|
81
48
|
}[];
|
|
82
49
|
_meta: {
|
|
83
|
-
usage
|
|
50
|
+
usage: {
|
|
84
51
|
prompt_tokens: number;
|
|
85
52
|
completion_tokens: number;
|
|
86
53
|
total_tokens: number;
|
|
87
|
-
}
|
|
54
|
+
};
|
|
55
|
+
server_version: string;
|
|
56
|
+
mime: string;
|
|
57
|
+
} | {
|
|
58
|
+
usage?: undefined;
|
|
88
59
|
server_version: string;
|
|
89
60
|
mime: string;
|
|
90
61
|
};
|
|
91
62
|
}>;
|
|
92
|
-
/**
|
|
93
|
-
* Resolve a caller-supplied input image into a URL the chat-completions
|
|
94
|
-
* API accepts. Local file paths are sandboxed via
|
|
95
|
-
* `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
|
|
96
|
-
* `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
|
|
97
|
-
*/
|
|
98
|
-
export declare function resolveInputImage(ref: string): Promise<string>;
|
|
99
|
-
export declare function mimeFromExt(ext: string): string | null;
|
|
100
|
-
export declare function buildUserContent(prompt: string, inputImages?: string[]): Promise<string | OpenAI.Chat.Completions.ChatCompletionContentPart[]>;
|
|
@@ -1,15 +1,12 @@
|
|
|
1
1
|
import { promises as fs } from 'fs';
|
|
2
|
-
import
|
|
3
|
-
import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
2
|
+
import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
4
3
|
import { parseBase64DataUrl } from './fetch-utils.js';
|
|
4
|
+
import { buildUserContent } from './generate-image-input.js';
|
|
5
5
|
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
6
6
|
import { SERVER_VERSION } from '../version.js';
|
|
7
7
|
import { logger } from '../logger.js';
|
|
8
8
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
9
9
|
const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
|
|
10
|
-
// OpenRouter-documented aspect ratios (standard + extended). Extended are
|
|
11
|
-
// only honored by models that support them (e.g. gemini-3.1-flash-image),
|
|
12
|
-
// others fall back to the model's default.
|
|
13
10
|
const VALID_ASPECT_RATIOS = new Set([
|
|
14
11
|
'1:1',
|
|
15
12
|
'2:3',
|
|
@@ -32,9 +29,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
32
29
|
if (!prompt?.trim()) {
|
|
33
30
|
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
34
31
|
}
|
|
35
|
-
// Audit entry. Bypasses the normal log level so operators always see a
|
|
36
|
-
// record of cost-incurring operations. Prompt preview is hard-capped
|
|
37
|
-
// at 80 chars to avoid PII spillage in log aggregators.
|
|
38
32
|
logger.audit('generate_image.start', {
|
|
39
33
|
model: model || DEFAULT_MODEL,
|
|
40
34
|
prompt_preview: prompt.slice(0, 80),
|
|
@@ -43,15 +37,12 @@ export async function handleGenerateImage(request, openai) {
|
|
|
43
37
|
save_path: save_path ? 'provided' : 'none',
|
|
44
38
|
input_images_count: input_images?.length ?? 0,
|
|
45
39
|
});
|
|
46
|
-
// Validate optional shape fields early so callers get a clear error
|
|
47
|
-
// instead of a cryptic upstream 400.
|
|
48
40
|
if (aspect_ratio !== undefined && !VALID_ASPECT_RATIOS.has(aspect_ratio)) {
|
|
49
|
-
return
|
|
41
|
+
return invalidEnumError('aspect_ratio', aspect_ratio, VALID_ASPECT_RATIOS);
|
|
50
42
|
}
|
|
51
43
|
if (image_size !== undefined && !VALID_IMAGE_SIZES.has(image_size)) {
|
|
52
|
-
return
|
|
44
|
+
return invalidEnumError('image_size', image_size, VALID_IMAGE_SIZES);
|
|
53
45
|
}
|
|
54
|
-
// Fail-fast on unsafe paths BEFORE spending tokens.
|
|
55
46
|
let safePathResolved = null;
|
|
56
47
|
if (save_path) {
|
|
57
48
|
try {
|
|
@@ -64,9 +55,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
64
55
|
return toolErrorFrom(ErrorCode.INTERNAL, err);
|
|
65
56
|
}
|
|
66
57
|
}
|
|
67
|
-
// Build the user message. With no `input_images`, this is the original
|
|
68
|
-
// string content; with refs, it becomes a multimodal
|
|
69
|
-
// ChatCompletionContentPart[] (text preamble + one image_url per ref).
|
|
70
58
|
let content;
|
|
71
59
|
try {
|
|
72
60
|
content = await buildUserContent(prompt, input_images);
|
|
@@ -77,14 +65,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
77
65
|
}
|
|
78
66
|
return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_images');
|
|
79
67
|
}
|
|
80
|
-
// Assemble the request body. OpenRouter's image-generation guide
|
|
81
|
-
// requires:
|
|
82
|
-
// - `modalities: ["image", "text"]` so multimodal models (like
|
|
83
|
-
// Gemini) know to emit an image, not just text. Caller can
|
|
84
|
-
// override via the `modalities` field.
|
|
85
|
-
// - `image_config.{aspect_ratio, image_size}` for shape control.
|
|
86
|
-
// The OpenAI SDK doesn't type these fields, but passes unknown members
|
|
87
|
-
// through to the server, so we attach them via a typed cast.
|
|
88
68
|
const imageConfig = {};
|
|
89
69
|
if (aspect_ratio)
|
|
90
70
|
imageConfig.aspect_ratio = aspect_ratio;
|
|
@@ -93,7 +73,7 @@ export async function handleGenerateImage(request, openai) {
|
|
|
93
73
|
const body = {
|
|
94
74
|
model: model || DEFAULT_MODEL,
|
|
95
75
|
messages: [{ role: 'user', content }],
|
|
96
|
-
modalities: modalities
|
|
76
|
+
modalities: modalities?.length ? modalities : ['image', 'text'],
|
|
97
77
|
};
|
|
98
78
|
if (Object.keys(imageConfig).length > 0)
|
|
99
79
|
body.image_config = imageConfig;
|
|
@@ -101,10 +81,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
101
81
|
body.max_tokens = max_tokens;
|
|
102
82
|
let completion;
|
|
103
83
|
try {
|
|
104
|
-
// OpenRouter-specific `image_config` isn't in the OpenAI SDK's typings,
|
|
105
|
-
// but the SDK passes unknown fields straight through to the server.
|
|
106
|
-
// We never pass `stream: true`, so the response is always
|
|
107
|
-
// ChatCompletion.
|
|
108
84
|
completion = (await openai.chat.completions.create(body));
|
|
109
85
|
}
|
|
110
86
|
catch (err) {
|
|
@@ -116,8 +92,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
116
92
|
}
|
|
117
93
|
const base64 = extractBase64(message);
|
|
118
94
|
if (!base64) {
|
|
119
|
-
// Model talked but did not emit an image. Surface this as a distinct
|
|
120
|
-
// condition so callers don't treat chatter as a successful image.
|
|
121
95
|
const messageContent = message.content;
|
|
122
96
|
const text = typeof messageContent === 'string' ? messageContent : JSON.stringify(messageContent);
|
|
123
97
|
return toolError(ErrorCode.UPSTREAM_REFUSED, `Model returned no image. Text response: ${text.slice(0, 300)}`, {
|
|
@@ -127,113 +101,56 @@ export async function handleGenerateImage(request, openai) {
|
|
|
127
101
|
}
|
|
128
102
|
if (safePathResolved) {
|
|
129
103
|
try {
|
|
130
|
-
await fs.writeFile(safePathResolved,
|
|
104
|
+
await fs.writeFile(safePathResolved, base64.data, { encoding: 'base64' });
|
|
131
105
|
}
|
|
132
106
|
catch (err) {
|
|
133
107
|
return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
|
|
134
108
|
}
|
|
135
|
-
|
|
109
|
+
}
|
|
110
|
+
return buildImageSuccessResult(base64, completion.usage, safePathResolved ?? undefined);
|
|
111
|
+
}
|
|
112
|
+
function invalidEnumError(field, value, allowed) {
|
|
113
|
+
return toolError(ErrorCode.INVALID_INPUT, `${field} '${value}' is not supported. Valid values: ${[...allowed].join(', ')}.`);
|
|
114
|
+
}
|
|
115
|
+
function buildImageSuccessResult(base64, usage, savePath) {
|
|
116
|
+
const usageMeta = usage
|
|
117
|
+
? {
|
|
118
|
+
usage: {
|
|
119
|
+
prompt_tokens: usage.prompt_tokens,
|
|
120
|
+
completion_tokens: usage.completion_tokens,
|
|
121
|
+
total_tokens: usage.total_tokens,
|
|
122
|
+
},
|
|
123
|
+
}
|
|
124
|
+
: {};
|
|
125
|
+
if (savePath) {
|
|
136
126
|
return {
|
|
137
127
|
content: [
|
|
138
|
-
{ type: 'text', text: `Image saved to: ${
|
|
128
|
+
{ type: 'text', text: `Image saved to: ${savePath}` },
|
|
139
129
|
{ type: 'image', mimeType: base64.mime, data: base64.data },
|
|
140
130
|
],
|
|
141
131
|
_meta: {
|
|
142
132
|
server_version: SERVER_VERSION,
|
|
143
|
-
save_path:
|
|
133
|
+
save_path: savePath,
|
|
144
134
|
mime: base64.mime,
|
|
145
|
-
...
|
|
146
|
-
? {
|
|
147
|
-
usage: {
|
|
148
|
-
prompt_tokens: usage.prompt_tokens,
|
|
149
|
-
completion_tokens: usage.completion_tokens,
|
|
150
|
-
total_tokens: usage.total_tokens,
|
|
151
|
-
},
|
|
152
|
-
}
|
|
153
|
-
: {}),
|
|
135
|
+
...usageMeta,
|
|
154
136
|
},
|
|
155
137
|
};
|
|
156
138
|
}
|
|
157
|
-
const usage = completion.usage;
|
|
158
139
|
return {
|
|
159
140
|
content: [{ type: 'image', mimeType: base64.mime, data: base64.data }],
|
|
160
141
|
_meta: {
|
|
161
142
|
server_version: SERVER_VERSION,
|
|
162
143
|
mime: base64.mime,
|
|
163
|
-
...
|
|
164
|
-
? {
|
|
165
|
-
usage: {
|
|
166
|
-
prompt_tokens: usage.prompt_tokens,
|
|
167
|
-
completion_tokens: usage.completion_tokens,
|
|
168
|
-
total_tokens: usage.total_tokens,
|
|
169
|
-
},
|
|
170
|
-
}
|
|
171
|
-
: {}),
|
|
144
|
+
...usageMeta,
|
|
172
145
|
},
|
|
173
146
|
};
|
|
174
147
|
}
|
|
175
|
-
/**
|
|
176
|
-
* Resolve a caller-supplied input image into a URL the chat-completions
|
|
177
|
-
* API accepts. Local file paths are sandboxed via
|
|
178
|
-
* `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
|
|
179
|
-
* `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
|
|
180
|
-
*/
|
|
181
|
-
export async function resolveInputImage(ref) {
|
|
182
|
-
const trimmed = ref.trim();
|
|
183
|
-
if (!trimmed)
|
|
184
|
-
throw new Error('empty input_images entry');
|
|
185
|
-
if (trimmed.startsWith('data:'))
|
|
186
|
-
return trimmed;
|
|
187
|
-
if (/^https?:\/\//i.test(trimmed))
|
|
188
|
-
return trimmed;
|
|
189
|
-
const abs = await resolveSafeInputPath(trimmed);
|
|
190
|
-
const buf = await fs.readFile(abs);
|
|
191
|
-
const mime = mimeFromExt(path.extname(abs)) || 'image/png';
|
|
192
|
-
return `data:${mime};base64,${buf.toString('base64')}`;
|
|
193
|
-
}
|
|
194
|
-
export function mimeFromExt(ext) {
|
|
195
|
-
const e = ext.toLowerCase().replace(/^\./, '');
|
|
196
|
-
switch (e) {
|
|
197
|
-
case 'png':
|
|
198
|
-
return 'image/png';
|
|
199
|
-
case 'jpg':
|
|
200
|
-
case 'jpeg':
|
|
201
|
-
return 'image/jpeg';
|
|
202
|
-
case 'webp':
|
|
203
|
-
return 'image/webp';
|
|
204
|
-
case 'gif':
|
|
205
|
-
return 'image/gif';
|
|
206
|
-
default:
|
|
207
|
-
return null;
|
|
208
|
-
}
|
|
209
|
-
}
|
|
210
|
-
export async function buildUserContent(prompt, inputImages) {
|
|
211
|
-
if (!inputImages?.length) {
|
|
212
|
-
return `Generate an image: ${prompt}`;
|
|
213
|
-
}
|
|
214
|
-
const parts = [
|
|
215
|
-
{
|
|
216
|
-
type: 'text',
|
|
217
|
-
text: `Generate an image based on this prompt, using the following reference image(s) ` +
|
|
218
|
-
`for visual consistency. Match the appearance, identity, and style of the references ` +
|
|
219
|
-
`closely; do not alter them.\n\nPrompt: ${prompt}`,
|
|
220
|
-
},
|
|
221
|
-
];
|
|
222
|
-
for (const ref of inputImages) {
|
|
223
|
-
const url = await resolveInputImage(ref);
|
|
224
|
-
parts.push({
|
|
225
|
-
type: 'image_url',
|
|
226
|
-
image_url: { url, detail: 'high' },
|
|
227
|
-
});
|
|
228
|
-
}
|
|
229
|
-
return parts;
|
|
230
|
-
}
|
|
231
148
|
function extractBase64(message) {
|
|
232
149
|
const images = message.images;
|
|
233
150
|
if (Array.isArray(images) && images.length) {
|
|
234
151
|
for (const img of images) {
|
|
235
152
|
const imageUrl = img.image_url;
|
|
236
|
-
const result =
|
|
153
|
+
const result = dataUrlToBase64(imageUrl?.url || img.url);
|
|
237
154
|
if (result)
|
|
238
155
|
return result;
|
|
239
156
|
}
|
|
@@ -243,7 +160,7 @@ function extractBase64(message) {
|
|
|
243
160
|
const iu = part.image_url;
|
|
244
161
|
const url = iu?.url || part.url;
|
|
245
162
|
if (url) {
|
|
246
|
-
const r =
|
|
163
|
+
const r = dataUrlToBase64(url);
|
|
247
164
|
if (r)
|
|
248
165
|
return r;
|
|
249
166
|
}
|
|
@@ -257,24 +174,20 @@ function extractBase64(message) {
|
|
|
257
174
|
}
|
|
258
175
|
}
|
|
259
176
|
if (typeof message.content === 'string') {
|
|
260
|
-
//
|
|
261
|
-
// a single regex here because data URLs may carry MIME parameters
|
|
262
|
-
// (e.g. `data:image/png;charset=binary;base64,...`) which trips the
|
|
263
|
-
// naive `data:([^;]+);base64,(.+)` form.
|
|
177
|
+
// Data URLs may carry MIME parameters (e.g. charset=binary) that break naive regexes.
|
|
264
178
|
const start = message.content.indexOf('data:image/');
|
|
265
179
|
if (start >= 0) {
|
|
266
|
-
// Find the end of the data URL: a whitespace or closing quote/paren.
|
|
267
180
|
const tail = message.content.slice(start);
|
|
268
181
|
const end = tail.search(/[\s)"']/);
|
|
269
182
|
const url = end === -1 ? tail : tail.slice(0, end);
|
|
270
|
-
const parsed =
|
|
183
|
+
const parsed = dataUrlToBase64(url);
|
|
271
184
|
if (parsed)
|
|
272
185
|
return parsed;
|
|
273
186
|
}
|
|
274
187
|
}
|
|
275
188
|
return null;
|
|
276
189
|
}
|
|
277
|
-
function
|
|
190
|
+
function dataUrlToBase64(url) {
|
|
278
191
|
if (!url?.startsWith('data:'))
|
|
279
192
|
return null;
|
|
280
193
|
const parsed = parseBase64DataUrl(url);
|
|
@@ -34,7 +34,7 @@ export declare function handleGenerateVideo(request: {
|
|
|
34
34
|
};
|
|
35
35
|
}, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
|
|
36
36
|
content: {
|
|
37
|
-
type:
|
|
37
|
+
type: string;
|
|
38
38
|
text: string;
|
|
39
39
|
}[];
|
|
40
40
|
isError: false;
|
|
@@ -97,7 +97,7 @@ export declare function handleGenerateVideoFromImage(request: {
|
|
|
97
97
|
};
|
|
98
98
|
}, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
|
|
99
99
|
content: {
|
|
100
|
-
type:
|
|
100
|
+
type: string;
|
|
101
101
|
text: string;
|
|
102
102
|
}[];
|
|
103
103
|
isError: false;
|
|
@@ -11,6 +11,34 @@ const DEFAULT_POLL_INTERVAL_MS = 15_000;
|
|
|
11
11
|
const DEFAULT_MAX_WAIT_MS = 10 * 60_000;
|
|
12
12
|
const MIN_POLL_INTERVAL_MS = 50; // just to avoid a 0ms busy-loop if a caller omits
|
|
13
13
|
const INLINE_RETURN_CEILING_BYTES = 10 * 1024 * 1024;
|
|
14
|
+
/** Models deprecated by OpenAI — removal date: 2026-09-24. */
|
|
15
|
+
const SORA_DEPRECATED_MODELS = new Set([
|
|
16
|
+
'openai/sora-2',
|
|
17
|
+
'openai/sora-2-pro',
|
|
18
|
+
'openai/sora-2-2025-10-06',
|
|
19
|
+
'openai/sora-2-2025-12-08',
|
|
20
|
+
'openai/sora-2-pro-2025-10-06',
|
|
21
|
+
]);
|
|
22
|
+
const SORA_ALTERNATIVES = [
|
|
23
|
+
'google/veo-3.1 (recommended — fast, audio support)',
|
|
24
|
+
'google/veo-3.1-fast (budget-friendly)',
|
|
25
|
+
'bytedance/seedance-2.0 (high quality)',
|
|
26
|
+
'bytedance/seedance-2.0-fast (fast turnaround)',
|
|
27
|
+
'alibaba/wan-2.7 (good for artistic styles)',
|
|
28
|
+
];
|
|
29
|
+
/**
|
|
30
|
+
* Check if the model is a deprecated Sora model and return a warning string,
|
|
31
|
+
* or null if no deprecation applies.
|
|
32
|
+
*/
|
|
33
|
+
function checkSoraDeprecation(model) {
|
|
34
|
+
const normalized = model.toLowerCase().trim();
|
|
35
|
+
if (!SORA_DEPRECATED_MODELS.has(normalized) && !normalized.startsWith('openai/sora')) {
|
|
36
|
+
return null;
|
|
37
|
+
}
|
|
38
|
+
return (`⚠️ DEPRECATION WARNING: ${model} is deprecated by OpenAI and will be removed from the API on September 24, 2026. ` +
|
|
39
|
+
`Your request will still be attempted, but may fail. Recommended alternatives:\n` +
|
|
40
|
+
SORA_ALTERNATIVES.map((a) => ` • ${a}`).join('\n'));
|
|
41
|
+
}
|
|
14
42
|
function getMaxInlineBytes() {
|
|
15
43
|
return readEnvInt('OPENROUTER_VIDEO_INLINE_MAX_BYTES', INLINE_RETURN_CEILING_BYTES, 4096);
|
|
16
44
|
}
|
|
@@ -88,34 +116,45 @@ function buildRequestBody(args, model) {
|
|
|
88
116
|
return body;
|
|
89
117
|
}
|
|
90
118
|
async function attachFrameImages(args, body) {
|
|
91
|
-
const
|
|
119
|
+
const frameTasks = [];
|
|
92
120
|
if (args.first_frame_image) {
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
121
|
+
frameTasks.push(prepareImageInput(args.first_frame_image).then((img) => img
|
|
122
|
+
? {
|
|
123
|
+
kind: 'frame',
|
|
124
|
+
entry: {
|
|
125
|
+
type: 'image_url',
|
|
126
|
+
image_url: { url: `data:${img.mime};base64,${img.data}` },
|
|
127
|
+
frame_type: 'first_frame',
|
|
128
|
+
},
|
|
129
|
+
}
|
|
130
|
+
: null));
|
|
100
131
|
}
|
|
101
132
|
if (args.last_frame_image) {
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
133
|
+
frameTasks.push(prepareImageInput(args.last_frame_image).then((img) => img
|
|
134
|
+
? {
|
|
135
|
+
kind: 'frame',
|
|
136
|
+
entry: {
|
|
137
|
+
type: 'image_url',
|
|
138
|
+
image_url: { url: `data:${img.mime};base64,${img.data}` },
|
|
139
|
+
frame_type: 'last_frame',
|
|
140
|
+
},
|
|
141
|
+
}
|
|
142
|
+
: null));
|
|
109
143
|
}
|
|
144
|
+
const frameResults = await Promise.all(frameTasks);
|
|
145
|
+
const frameImages = frameResults
|
|
146
|
+
.filter((r) => r !== null)
|
|
147
|
+
.map((r) => r.entry);
|
|
110
148
|
if (frameImages.length)
|
|
111
149
|
body.frame_images = frameImages;
|
|
112
150
|
if (args.reference_images?.length) {
|
|
113
|
-
const
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
151
|
+
const refResults = await Promise.all(args.reference_images.map((src) => prepareImageInput(src)));
|
|
152
|
+
const refs = refResults
|
|
153
|
+
.filter((img) => img !== null)
|
|
154
|
+
.map((img) => ({
|
|
155
|
+
type: 'image_url',
|
|
156
|
+
image_url: { url: `data:${img.mime};base64,${img.data}` },
|
|
157
|
+
}));
|
|
119
158
|
if (refs.length)
|
|
120
159
|
body.input_references = refs;
|
|
121
160
|
}
|
|
@@ -234,9 +273,10 @@ export async function handleGenerateVideo(request, apiClient, progress) {
|
|
|
234
273
|
if (!args.prompt || !args.prompt.trim()) {
|
|
235
274
|
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
236
275
|
}
|
|
237
|
-
const model = args.model ||
|
|
238
|
-
|
|
239
|
-
|
|
276
|
+
const model = args.model || process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL || FALLBACK_MODEL;
|
|
277
|
+
// Sora deprecation warning — OpenAI is removing the Videos API and all
|
|
278
|
+
// Sora 2 model aliases on September 24, 2026. Warn and suggest alternatives.
|
|
279
|
+
const deprecationWarning = checkSoraDeprecation(model);
|
|
240
280
|
// Audit entry — video is the most expensive tool we have. Always log
|
|
241
281
|
// model, resolution, duration, and a safe prompt preview so unintended
|
|
242
282
|
// spend can be traced.
|
|
@@ -297,13 +337,16 @@ export async function handleGenerateVideo(request, apiClient, progress) {
|
|
|
297
337
|
});
|
|
298
338
|
}
|
|
299
339
|
if (outcome.kind === 'timeout') {
|
|
340
|
+
const timeoutContent = [];
|
|
341
|
+
if (deprecationWarning) {
|
|
342
|
+
timeoutContent.push({ type: 'text', text: deprecationWarning });
|
|
343
|
+
}
|
|
344
|
+
timeoutContent.push({
|
|
345
|
+
type: 'text',
|
|
346
|
+
text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
|
|
347
|
+
});
|
|
300
348
|
return {
|
|
301
|
-
content:
|
|
302
|
-
{
|
|
303
|
-
type: 'text',
|
|
304
|
-
text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
|
|
305
|
-
},
|
|
306
|
-
],
|
|
349
|
+
content: timeoutContent,
|
|
307
350
|
isError: false,
|
|
308
351
|
_meta: {
|
|
309
352
|
server_version: SERVER_VERSION,
|
|
@@ -316,6 +359,11 @@ export async function handleGenerateVideo(request, apiClient, progress) {
|
|
|
316
359
|
}
|
|
317
360
|
try {
|
|
318
361
|
const { content, _meta } = await finalizeCompletedJob(apiClient, outcome.status, safeSavePath);
|
|
362
|
+
// Prepend deprecation warning if applicable
|
|
363
|
+
if (deprecationWarning) {
|
|
364
|
+
content.unshift({ type: 'text', text: deprecationWarning });
|
|
365
|
+
_meta.deprecated_model = true;
|
|
366
|
+
}
|
|
319
367
|
return { content, _meta };
|
|
320
368
|
}
|
|
321
369
|
catch (err) {
|
|
@@ -3,6 +3,7 @@ export declare const isBlockedIPv4: typeof _isBlockedIPv4;
|
|
|
3
3
|
export declare const assertUrlSafeForFetch: typeof _assertUrlSafeForFetch;
|
|
4
4
|
export declare function getMaxImageDimension(): number;
|
|
5
5
|
export declare function getImageJpegQuality(): number;
|
|
6
|
+
export declare function mimeFromExtension(ext: string): string | null;
|
|
6
7
|
export declare function getMimeType(filePath: string): string;
|
|
7
8
|
export declare function fetchHttpImage(urlString: string): Promise<Buffer>;
|
|
8
9
|
export declare function fetchImage(source: string): Promise<Buffer>;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import path from 'path';
|
|
2
2
|
import { promises as fs } from 'fs';
|
|
3
3
|
import { readEnvInt, isBlockedIPv4 as _isBlockedIPv4, assertUrlSafeForFetch as _assertUrlSafeForFetch, fetchHttpResource, parseBase64DataUrl, } from './fetch-utils.js';
|
|
4
|
+
import { resolveSafeInputPath } from './path-safety.js';
|
|
4
5
|
// Re-export for backward compatibility (tests import from image-utils)
|
|
5
6
|
export const isBlockedIPv4 = _isBlockedIPv4;
|
|
6
7
|
export const assertUrlSafeForFetch = _assertUrlSafeForFetch;
|
|
@@ -39,22 +40,29 @@ async function loadSharp() {
|
|
|
39
40
|
sharpFn = fn ?? mod;
|
|
40
41
|
}
|
|
41
42
|
catch {
|
|
42
|
-
|
|
43
|
+
// sharp is optional — images will be sent unprocessed (larger but functional)
|
|
44
|
+
const { logger } = await import('../logger.js');
|
|
45
|
+
logger.warn('sharp not available, images will be sent unprocessed');
|
|
43
46
|
}
|
|
44
47
|
}
|
|
45
48
|
return sharpFn;
|
|
46
49
|
}
|
|
50
|
+
const IMAGE_EXT_MIME = {
|
|
51
|
+
png: 'image/png',
|
|
52
|
+
jpg: 'image/jpeg',
|
|
53
|
+
jpeg: 'image/jpeg',
|
|
54
|
+
webp: 'image/webp',
|
|
55
|
+
gif: 'image/gif',
|
|
56
|
+
bmp: 'image/bmp',
|
|
57
|
+
};
|
|
58
|
+
export function mimeFromExtension(ext) {
|
|
59
|
+
const normalized = ext.toLowerCase().replace(/^\./, '');
|
|
60
|
+
if (!normalized)
|
|
61
|
+
return null;
|
|
62
|
+
return IMAGE_EXT_MIME[normalized] ?? null;
|
|
63
|
+
}
|
|
47
64
|
export function getMimeType(filePath) {
|
|
48
|
-
|
|
49
|
-
const map = {
|
|
50
|
-
'.png': 'image/png',
|
|
51
|
-
'.jpg': 'image/jpeg',
|
|
52
|
-
'.jpeg': 'image/jpeg',
|
|
53
|
-
'.webp': 'image/webp',
|
|
54
|
-
'.gif': 'image/gif',
|
|
55
|
-
'.bmp': 'image/bmp',
|
|
56
|
-
};
|
|
57
|
-
return map[ext] || 'image/jpeg';
|
|
65
|
+
return mimeFromExtension(path.extname(filePath)) ?? 'image/jpeg';
|
|
58
66
|
}
|
|
59
67
|
export async function fetchHttpImage(urlString) {
|
|
60
68
|
const { buffer } = await fetchHttpResource(urlString, {
|
|
@@ -77,7 +85,8 @@ export async function fetchImage(source) {
|
|
|
77
85
|
if (source.startsWith('http://') || source.startsWith('https://')) {
|
|
78
86
|
return fetchHttpImage(source);
|
|
79
87
|
}
|
|
80
|
-
|
|
88
|
+
const safe = await resolveSafeInputPath(source);
|
|
89
|
+
return fs.readFile(safe);
|
|
81
90
|
}
|
|
82
91
|
/**
|
|
83
92
|
* Sniff image MIME type from magic bytes. Used to label the output of a
|
|
@@ -134,13 +143,14 @@ export async function optimizeImage(buffer) {
|
|
|
134
143
|
const maxDim = getMaxImageDimension();
|
|
135
144
|
const quality = getImageJpegQuality();
|
|
136
145
|
try {
|
|
137
|
-
const
|
|
138
|
-
|
|
146
|
+
const pipeline = sharp(buffer);
|
|
147
|
+
const meta = await pipeline.metadata();
|
|
148
|
+
let resized = pipeline;
|
|
139
149
|
if (meta.width && meta.height && Math.max(meta.width, meta.height) > maxDim) {
|
|
140
150
|
const opts = meta.width > meta.height ? { width: maxDim } : { height: maxDim };
|
|
141
|
-
|
|
151
|
+
resized = pipeline.resize(opts);
|
|
142
152
|
}
|
|
143
|
-
const out = await
|
|
153
|
+
const out = await resized.jpeg({ quality }).toBuffer();
|
|
144
154
|
return { base64: out.toString('base64'), mime: 'image/jpeg' };
|
|
145
155
|
}
|
|
146
156
|
catch {
|
|
@@ -110,7 +110,9 @@ export function classifyUpstreamError(err, contextMessage) {
|
|
|
110
110
|
}
|
|
111
111
|
// Model lookup failures.
|
|
112
112
|
if (lower.includes('model') &&
|
|
113
|
-
(lower.includes('does not exist') ||
|
|
113
|
+
(lower.includes('does not exist') ||
|
|
114
|
+
lower.includes('not found') ||
|
|
115
|
+
lower.includes('invalid model'))) {
|
|
114
116
|
return toolError(ErrorCode.MODEL_NOT_FOUND, fullMsg, { status }, {
|
|
115
117
|
suggestions: [
|
|
116
118
|
'Use search_models to discover valid model ids',
|
|
@@ -119,7 +121,9 @@ export function classifyUpstreamError(err, contextMessage) {
|
|
|
119
121
|
});
|
|
120
122
|
}
|
|
121
123
|
// Content policy / moderation — surface as UPSTREAM_REFUSED so callers can distinguish from 5xx.
|
|
122
|
-
if (lower.includes('content policy') ||
|
|
124
|
+
if (lower.includes('content policy') ||
|
|
125
|
+
lower.includes('moderation') ||
|
|
126
|
+
lower.includes('refused')) {
|
|
123
127
|
return toolError(ErrorCode.UPSTREAM_REFUSED, fullMsg, { status, reason: 'policy' }, {
|
|
124
128
|
suggestions: ['Rephrase the prompt', 'Try a different provider via provider.order'],
|
|
125
129
|
});
|