@stabgan/openrouter-mcp-multimodal 4.5.0 → 4.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +380 -242
- package/dist/index.js +22 -6
- package/dist/model-cache.d.ts +35 -12
- package/dist/model-cache.js +79 -22
- package/dist/tool-descriptions.d.ts +19 -0
- package/dist/tool-descriptions.js +423 -0
- package/dist/tool-handlers/analyze-audio.js +5 -1
- package/dist/tool-handlers/analyze-image.js +6 -5
- package/dist/tool-handlers/analyze-video.js +6 -5
- package/dist/tool-handlers/audio-utils.js +4 -2
- package/dist/tool-handlers/chat-completion.js +1 -1
- package/dist/tool-handlers/fetch-utils.js +16 -2
- package/dist/tool-handlers/generate-audio.js +2 -4
- package/dist/tool-handlers/generate-image-input.d.ts +3 -0
- package/dist/tool-handlers/generate-image-input.js +38 -0
- package/dist/tool-handlers/generate-image.d.ts +13 -51
- package/dist/tool-handlers/generate-image.js +32 -119
- package/dist/tool-handlers/generate-video.js +28 -24
- package/dist/tool-handlers/health-check.js +4 -1
- package/dist/tool-handlers/image-utils.d.ts +1 -0
- package/dist/tool-handlers/image-utils.js +26 -16
- package/dist/tool-handlers/openrouter-errors.d.ts +5 -1
- package/dist/tool-handlers/openrouter-errors.js +78 -13
- package/dist/tool-handlers/provider-routing.js +7 -2
- package/dist/tool-handlers/rerank.js +2 -5
- package/dist/tool-handlers/search-models.d.ts +2 -2
- package/dist/tool-handlers/search-models.js +2 -6
- package/dist/tool-handlers/structured-output.d.ts +8 -0
- package/dist/tool-handlers/structured-output.js +11 -0
- package/dist/tool-handlers/video-utils.js +6 -9
- package/dist/tool-handlers.js +43 -123
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +26 -14
|
@@ -3,48 +3,10 @@ export interface GenerateImageToolRequest {
|
|
|
3
3
|
prompt: string;
|
|
4
4
|
model?: string;
|
|
5
5
|
save_path?: string;
|
|
6
|
-
/**
|
|
7
|
-
* Output aspect ratio. Passed through as `image_config.aspect_ratio`.
|
|
8
|
-
* Supported by OpenRouter image models (e.g. `1:1`, `16:9`, `9:16`,
|
|
9
|
-
* `4:3`, `3:4`, `21:9`). Model-dependent. Unsupported values fall back
|
|
10
|
-
* to the model's default. See
|
|
11
|
-
* https://openrouter.ai/docs/guides/overview/multimodal/image-generation
|
|
12
|
-
*/
|
|
13
6
|
aspect_ratio?: string;
|
|
14
|
-
/**
|
|
15
|
-
* Output image resolution bucket. Passed through as
|
|
16
|
-
* `image_config.image_size`. Typical values: `0.5K`, `1K` (default),
|
|
17
|
-
* `2K`, `4K`. Model-dependent.
|
|
18
|
-
*/
|
|
19
7
|
image_size?: string;
|
|
20
|
-
/**
|
|
21
|
-
* Upper bound on the completion budget. Without this OpenRouter
|
|
22
|
-
* reserves the model's full context window (~29k for Gemini
|
|
23
|
-
* image models), which can trigger a 402 on low-credit accounts even
|
|
24
|
-
* though the actual generation uses far fewer tokens. 4096 is plenty
|
|
25
|
-
* for the image payload + any caption.
|
|
26
|
-
*/
|
|
27
8
|
max_tokens?: number;
|
|
28
|
-
/**
|
|
29
|
-
* Optional reference images. Each entry is one of:
|
|
30
|
-
* - a `data:image/...;base64,...` URL,
|
|
31
|
-
* - an `http(s)://` URL (OpenRouter fetches it),
|
|
32
|
-
* - a local file path (sandboxed to `OPENROUTER_INPUT_DIR` /
|
|
33
|
-
* `OPENROUTER_OUTPUT_DIR` / cwd, read + base64-encoded by the server).
|
|
34
|
-
*
|
|
35
|
-
* When provided, the user message becomes multimodal: a text prompt
|
|
36
|
-
* plus one `image_url` block per reference, in array order. Enables
|
|
37
|
-
* character / style consistency and image-to-image refinement on
|
|
38
|
-
* chat-image models that accept image inputs (Gemini Nano Banana,
|
|
39
|
-
* `openai/gpt-5.4-image-2`).
|
|
40
|
-
*/
|
|
41
9
|
input_images?: string[];
|
|
42
|
-
/**
|
|
43
|
-
* Override the default `modalities: ["image","text"]` sent to
|
|
44
|
-
* OpenRouter. Most callers should leave this unset. Provide e.g.
|
|
45
|
-
* `["text"]` to suppress image output for inspection / captioning,
|
|
46
|
-
* or other shapes for future model variants.
|
|
47
|
-
*/
|
|
48
10
|
modalities?: string[];
|
|
49
11
|
}
|
|
50
12
|
export declare function handleGenerateImage(request: {
|
|
@@ -64,11 +26,16 @@ export declare function handleGenerateImage(request: {
|
|
|
64
26
|
text?: undefined;
|
|
65
27
|
})[];
|
|
66
28
|
_meta: {
|
|
67
|
-
usage
|
|
29
|
+
usage: {
|
|
68
30
|
prompt_tokens: number;
|
|
69
31
|
completion_tokens: number;
|
|
70
32
|
total_tokens: number;
|
|
71
|
-
}
|
|
33
|
+
};
|
|
34
|
+
server_version: string;
|
|
35
|
+
save_path: string;
|
|
36
|
+
mime: string;
|
|
37
|
+
} | {
|
|
38
|
+
usage?: undefined;
|
|
72
39
|
server_version: string;
|
|
73
40
|
save_path: string;
|
|
74
41
|
mime: string;
|
|
@@ -80,21 +47,16 @@ export declare function handleGenerateImage(request: {
|
|
|
80
47
|
data: string;
|
|
81
48
|
}[];
|
|
82
49
|
_meta: {
|
|
83
|
-
usage
|
|
50
|
+
usage: {
|
|
84
51
|
prompt_tokens: number;
|
|
85
52
|
completion_tokens: number;
|
|
86
53
|
total_tokens: number;
|
|
87
|
-
}
|
|
54
|
+
};
|
|
55
|
+
server_version: string;
|
|
56
|
+
mime: string;
|
|
57
|
+
} | {
|
|
58
|
+
usage?: undefined;
|
|
88
59
|
server_version: string;
|
|
89
60
|
mime: string;
|
|
90
61
|
};
|
|
91
62
|
}>;
|
|
92
|
-
/**
|
|
93
|
-
* Resolve a caller-supplied input image into a URL the chat-completions
|
|
94
|
-
* API accepts. Local file paths are sandboxed via
|
|
95
|
-
* `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
|
|
96
|
-
* `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
|
|
97
|
-
*/
|
|
98
|
-
export declare function resolveInputImage(ref: string): Promise<string>;
|
|
99
|
-
export declare function mimeFromExt(ext: string): string | null;
|
|
100
|
-
export declare function buildUserContent(prompt: string, inputImages?: string[]): Promise<string | OpenAI.Chat.Completions.ChatCompletionContentPart[]>;
|
|
@@ -1,15 +1,12 @@
|
|
|
1
1
|
import { promises as fs } from 'fs';
|
|
2
|
-
import
|
|
3
|
-
import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
2
|
+
import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
4
3
|
import { parseBase64DataUrl } from './fetch-utils.js';
|
|
4
|
+
import { buildUserContent } from './generate-image-input.js';
|
|
5
5
|
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
6
6
|
import { SERVER_VERSION } from '../version.js';
|
|
7
7
|
import { logger } from '../logger.js';
|
|
8
8
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
9
9
|
const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
|
|
10
|
-
// OpenRouter-documented aspect ratios (standard + extended). Extended are
|
|
11
|
-
// only honored by models that support them (e.g. gemini-3.1-flash-image),
|
|
12
|
-
// others fall back to the model's default.
|
|
13
10
|
const VALID_ASPECT_RATIOS = new Set([
|
|
14
11
|
'1:1',
|
|
15
12
|
'2:3',
|
|
@@ -32,9 +29,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
32
29
|
if (!prompt?.trim()) {
|
|
33
30
|
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
34
31
|
}
|
|
35
|
-
// Audit entry. Bypasses the normal log level so operators always see a
|
|
36
|
-
// record of cost-incurring operations. Prompt preview is hard-capped
|
|
37
|
-
// at 80 chars to avoid PII spillage in log aggregators.
|
|
38
32
|
logger.audit('generate_image.start', {
|
|
39
33
|
model: model || DEFAULT_MODEL,
|
|
40
34
|
prompt_preview: prompt.slice(0, 80),
|
|
@@ -43,15 +37,12 @@ export async function handleGenerateImage(request, openai) {
|
|
|
43
37
|
save_path: save_path ? 'provided' : 'none',
|
|
44
38
|
input_images_count: input_images?.length ?? 0,
|
|
45
39
|
});
|
|
46
|
-
// Validate optional shape fields early so callers get a clear error
|
|
47
|
-
// instead of a cryptic upstream 400.
|
|
48
40
|
if (aspect_ratio !== undefined && !VALID_ASPECT_RATIOS.has(aspect_ratio)) {
|
|
49
|
-
return
|
|
41
|
+
return invalidEnumError('aspect_ratio', aspect_ratio, VALID_ASPECT_RATIOS);
|
|
50
42
|
}
|
|
51
43
|
if (image_size !== undefined && !VALID_IMAGE_SIZES.has(image_size)) {
|
|
52
|
-
return
|
|
44
|
+
return invalidEnumError('image_size', image_size, VALID_IMAGE_SIZES);
|
|
53
45
|
}
|
|
54
|
-
// Fail-fast on unsafe paths BEFORE spending tokens.
|
|
55
46
|
let safePathResolved = null;
|
|
56
47
|
if (save_path) {
|
|
57
48
|
try {
|
|
@@ -64,9 +55,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
64
55
|
return toolErrorFrom(ErrorCode.INTERNAL, err);
|
|
65
56
|
}
|
|
66
57
|
}
|
|
67
|
-
// Build the user message. With no `input_images`, this is the original
|
|
68
|
-
// string content; with refs, it becomes a multimodal
|
|
69
|
-
// ChatCompletionContentPart[] (text preamble + one image_url per ref).
|
|
70
58
|
let content;
|
|
71
59
|
try {
|
|
72
60
|
content = await buildUserContent(prompt, input_images);
|
|
@@ -77,14 +65,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
77
65
|
}
|
|
78
66
|
return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_images');
|
|
79
67
|
}
|
|
80
|
-
// Assemble the request body. OpenRouter's image-generation guide
|
|
81
|
-
// requires:
|
|
82
|
-
// - `modalities: ["image", "text"]` so multimodal models (like
|
|
83
|
-
// Gemini) know to emit an image, not just text. Caller can
|
|
84
|
-
// override via the `modalities` field.
|
|
85
|
-
// - `image_config.{aspect_ratio, image_size}` for shape control.
|
|
86
|
-
// The OpenAI SDK doesn't type these fields, but passes unknown members
|
|
87
|
-
// through to the server, so we attach them via a typed cast.
|
|
88
68
|
const imageConfig = {};
|
|
89
69
|
if (aspect_ratio)
|
|
90
70
|
imageConfig.aspect_ratio = aspect_ratio;
|
|
@@ -93,7 +73,7 @@ export async function handleGenerateImage(request, openai) {
|
|
|
93
73
|
const body = {
|
|
94
74
|
model: model || DEFAULT_MODEL,
|
|
95
75
|
messages: [{ role: 'user', content }],
|
|
96
|
-
modalities: modalities
|
|
76
|
+
modalities: modalities?.length ? modalities : ['image', 'text'],
|
|
97
77
|
};
|
|
98
78
|
if (Object.keys(imageConfig).length > 0)
|
|
99
79
|
body.image_config = imageConfig;
|
|
@@ -101,10 +81,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
101
81
|
body.max_tokens = max_tokens;
|
|
102
82
|
let completion;
|
|
103
83
|
try {
|
|
104
|
-
// OpenRouter-specific `image_config` isn't in the OpenAI SDK's typings,
|
|
105
|
-
// but the SDK passes unknown fields straight through to the server.
|
|
106
|
-
// We never pass `stream: true`, so the response is always
|
|
107
|
-
// ChatCompletion.
|
|
108
84
|
completion = (await openai.chat.completions.create(body));
|
|
109
85
|
}
|
|
110
86
|
catch (err) {
|
|
@@ -116,8 +92,6 @@ export async function handleGenerateImage(request, openai) {
|
|
|
116
92
|
}
|
|
117
93
|
const base64 = extractBase64(message);
|
|
118
94
|
if (!base64) {
|
|
119
|
-
// Model talked but did not emit an image. Surface this as a distinct
|
|
120
|
-
// condition so callers don't treat chatter as a successful image.
|
|
121
95
|
const messageContent = message.content;
|
|
122
96
|
const text = typeof messageContent === 'string' ? messageContent : JSON.stringify(messageContent);
|
|
123
97
|
return toolError(ErrorCode.UPSTREAM_REFUSED, `Model returned no image. Text response: ${text.slice(0, 300)}`, {
|
|
@@ -127,113 +101,56 @@ export async function handleGenerateImage(request, openai) {
|
|
|
127
101
|
}
|
|
128
102
|
if (safePathResolved) {
|
|
129
103
|
try {
|
|
130
|
-
await fs.writeFile(safePathResolved,
|
|
104
|
+
await fs.writeFile(safePathResolved, base64.data, { encoding: 'base64' });
|
|
131
105
|
}
|
|
132
106
|
catch (err) {
|
|
133
107
|
return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
|
|
134
108
|
}
|
|
135
|
-
|
|
109
|
+
}
|
|
110
|
+
return buildImageSuccessResult(base64, completion.usage, safePathResolved ?? undefined);
|
|
111
|
+
}
|
|
112
|
+
function invalidEnumError(field, value, allowed) {
|
|
113
|
+
return toolError(ErrorCode.INVALID_INPUT, `${field} '${value}' is not supported. Valid values: ${[...allowed].join(', ')}.`);
|
|
114
|
+
}
|
|
115
|
+
function buildImageSuccessResult(base64, usage, savePath) {
|
|
116
|
+
const usageMeta = usage
|
|
117
|
+
? {
|
|
118
|
+
usage: {
|
|
119
|
+
prompt_tokens: usage.prompt_tokens,
|
|
120
|
+
completion_tokens: usage.completion_tokens,
|
|
121
|
+
total_tokens: usage.total_tokens,
|
|
122
|
+
},
|
|
123
|
+
}
|
|
124
|
+
: {};
|
|
125
|
+
if (savePath) {
|
|
136
126
|
return {
|
|
137
127
|
content: [
|
|
138
|
-
{ type: 'text', text: `Image saved to: ${
|
|
128
|
+
{ type: 'text', text: `Image saved to: ${savePath}` },
|
|
139
129
|
{ type: 'image', mimeType: base64.mime, data: base64.data },
|
|
140
130
|
],
|
|
141
131
|
_meta: {
|
|
142
132
|
server_version: SERVER_VERSION,
|
|
143
|
-
save_path:
|
|
133
|
+
save_path: savePath,
|
|
144
134
|
mime: base64.mime,
|
|
145
|
-
...
|
|
146
|
-
? {
|
|
147
|
-
usage: {
|
|
148
|
-
prompt_tokens: usage.prompt_tokens,
|
|
149
|
-
completion_tokens: usage.completion_tokens,
|
|
150
|
-
total_tokens: usage.total_tokens,
|
|
151
|
-
},
|
|
152
|
-
}
|
|
153
|
-
: {}),
|
|
135
|
+
...usageMeta,
|
|
154
136
|
},
|
|
155
137
|
};
|
|
156
138
|
}
|
|
157
|
-
const usage = completion.usage;
|
|
158
139
|
return {
|
|
159
140
|
content: [{ type: 'image', mimeType: base64.mime, data: base64.data }],
|
|
160
141
|
_meta: {
|
|
161
142
|
server_version: SERVER_VERSION,
|
|
162
143
|
mime: base64.mime,
|
|
163
|
-
...
|
|
164
|
-
? {
|
|
165
|
-
usage: {
|
|
166
|
-
prompt_tokens: usage.prompt_tokens,
|
|
167
|
-
completion_tokens: usage.completion_tokens,
|
|
168
|
-
total_tokens: usage.total_tokens,
|
|
169
|
-
},
|
|
170
|
-
}
|
|
171
|
-
: {}),
|
|
144
|
+
...usageMeta,
|
|
172
145
|
},
|
|
173
146
|
};
|
|
174
147
|
}
|
|
175
|
-
/**
|
|
176
|
-
* Resolve a caller-supplied input image into a URL the chat-completions
|
|
177
|
-
* API accepts. Local file paths are sandboxed via
|
|
178
|
-
* `resolveSafeInputPath` (`OPENROUTER_INPUT_DIR` /
|
|
179
|
-
* `OPENROUTER_OUTPUT_DIR` / cwd) and inlined as base64 data URLs.
|
|
180
|
-
*/
|
|
181
|
-
export async function resolveInputImage(ref) {
|
|
182
|
-
const trimmed = ref.trim();
|
|
183
|
-
if (!trimmed)
|
|
184
|
-
throw new Error('empty input_images entry');
|
|
185
|
-
if (trimmed.startsWith('data:'))
|
|
186
|
-
return trimmed;
|
|
187
|
-
if (/^https?:\/\//i.test(trimmed))
|
|
188
|
-
return trimmed;
|
|
189
|
-
const abs = await resolveSafeInputPath(trimmed);
|
|
190
|
-
const buf = await fs.readFile(abs);
|
|
191
|
-
const mime = mimeFromExt(path.extname(abs)) || 'image/png';
|
|
192
|
-
return `data:${mime};base64,${buf.toString('base64')}`;
|
|
193
|
-
}
|
|
194
|
-
export function mimeFromExt(ext) {
|
|
195
|
-
const e = ext.toLowerCase().replace(/^\./, '');
|
|
196
|
-
switch (e) {
|
|
197
|
-
case 'png':
|
|
198
|
-
return 'image/png';
|
|
199
|
-
case 'jpg':
|
|
200
|
-
case 'jpeg':
|
|
201
|
-
return 'image/jpeg';
|
|
202
|
-
case 'webp':
|
|
203
|
-
return 'image/webp';
|
|
204
|
-
case 'gif':
|
|
205
|
-
return 'image/gif';
|
|
206
|
-
default:
|
|
207
|
-
return null;
|
|
208
|
-
}
|
|
209
|
-
}
|
|
210
|
-
export async function buildUserContent(prompt, inputImages) {
|
|
211
|
-
if (!inputImages?.length) {
|
|
212
|
-
return `Generate an image: ${prompt}`;
|
|
213
|
-
}
|
|
214
|
-
const parts = [
|
|
215
|
-
{
|
|
216
|
-
type: 'text',
|
|
217
|
-
text: `Generate an image based on this prompt, using the following reference image(s) ` +
|
|
218
|
-
`for visual consistency. Match the appearance, identity, and style of the references ` +
|
|
219
|
-
`closely; do not alter them.\n\nPrompt: ${prompt}`,
|
|
220
|
-
},
|
|
221
|
-
];
|
|
222
|
-
for (const ref of inputImages) {
|
|
223
|
-
const url = await resolveInputImage(ref);
|
|
224
|
-
parts.push({
|
|
225
|
-
type: 'image_url',
|
|
226
|
-
image_url: { url, detail: 'high' },
|
|
227
|
-
});
|
|
228
|
-
}
|
|
229
|
-
return parts;
|
|
230
|
-
}
|
|
231
148
|
function extractBase64(message) {
|
|
232
149
|
const images = message.images;
|
|
233
150
|
if (Array.isArray(images) && images.length) {
|
|
234
151
|
for (const img of images) {
|
|
235
152
|
const imageUrl = img.image_url;
|
|
236
|
-
const result =
|
|
153
|
+
const result = dataUrlToBase64(imageUrl?.url || img.url);
|
|
237
154
|
if (result)
|
|
238
155
|
return result;
|
|
239
156
|
}
|
|
@@ -243,7 +160,7 @@ function extractBase64(message) {
|
|
|
243
160
|
const iu = part.image_url;
|
|
244
161
|
const url = iu?.url || part.url;
|
|
245
162
|
if (url) {
|
|
246
|
-
const r =
|
|
163
|
+
const r = dataUrlToBase64(url);
|
|
247
164
|
if (r)
|
|
248
165
|
return r;
|
|
249
166
|
}
|
|
@@ -257,24 +174,20 @@ function extractBase64(message) {
|
|
|
257
174
|
}
|
|
258
175
|
}
|
|
259
176
|
if (typeof message.content === 'string') {
|
|
260
|
-
//
|
|
261
|
-
// a single regex here because data URLs may carry MIME parameters
|
|
262
|
-
// (e.g. `data:image/png;charset=binary;base64,...`) which trips the
|
|
263
|
-
// naive `data:([^;]+);base64,(.+)` form.
|
|
177
|
+
// Data URLs may carry MIME parameters (e.g. charset=binary) that break naive regexes.
|
|
264
178
|
const start = message.content.indexOf('data:image/');
|
|
265
179
|
if (start >= 0) {
|
|
266
|
-
// Find the end of the data URL: a whitespace or closing quote/paren.
|
|
267
180
|
const tail = message.content.slice(start);
|
|
268
181
|
const end = tail.search(/[\s)"']/);
|
|
269
182
|
const url = end === -1 ? tail : tail.slice(0, end);
|
|
270
|
-
const parsed =
|
|
183
|
+
const parsed = dataUrlToBase64(url);
|
|
271
184
|
if (parsed)
|
|
272
185
|
return parsed;
|
|
273
186
|
}
|
|
274
187
|
}
|
|
275
188
|
return null;
|
|
276
189
|
}
|
|
277
|
-
function
|
|
190
|
+
function dataUrlToBase64(url) {
|
|
278
191
|
if (!url?.startsWith('data:'))
|
|
279
192
|
return null;
|
|
280
193
|
const parsed = parseBase64DataUrl(url);
|
|
@@ -88,34 +88,40 @@ function buildRequestBody(args, model) {
|
|
|
88
88
|
return body;
|
|
89
89
|
}
|
|
90
90
|
async function attachFrameImages(args, body) {
|
|
91
|
-
const
|
|
91
|
+
const frameTasks = [];
|
|
92
92
|
if (args.first_frame_image) {
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
93
|
+
frameTasks.push(prepareImageInput(args.first_frame_image).then((img) => img
|
|
94
|
+
? {
|
|
95
|
+
kind: 'frame',
|
|
96
|
+
entry: {
|
|
97
|
+
frame_type: 'first_frame',
|
|
98
|
+
image: { url: `data:${img.mime};base64,${img.data}` },
|
|
99
|
+
},
|
|
100
|
+
}
|
|
101
|
+
: null));
|
|
100
102
|
}
|
|
101
103
|
if (args.last_frame_image) {
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
104
|
+
frameTasks.push(prepareImageInput(args.last_frame_image).then((img) => img
|
|
105
|
+
? {
|
|
106
|
+
kind: 'frame',
|
|
107
|
+
entry: {
|
|
108
|
+
frame_type: 'last_frame',
|
|
109
|
+
image: { url: `data:${img.mime};base64,${img.data}` },
|
|
110
|
+
},
|
|
111
|
+
}
|
|
112
|
+
: null));
|
|
109
113
|
}
|
|
114
|
+
const frameResults = await Promise.all(frameTasks);
|
|
115
|
+
const frameImages = frameResults
|
|
116
|
+
.filter((r) => r !== null)
|
|
117
|
+
.map((r) => r.entry);
|
|
110
118
|
if (frameImages.length)
|
|
111
119
|
body.frame_images = frameImages;
|
|
112
120
|
if (args.reference_images?.length) {
|
|
113
|
-
const
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
refs.push({ image: { url: `data:${img.mime};base64,${img.data}` } });
|
|
118
|
-
}
|
|
121
|
+
const refResults = await Promise.all(args.reference_images.map((src) => prepareImageInput(src)));
|
|
122
|
+
const refs = refResults
|
|
123
|
+
.filter((img) => img !== null)
|
|
124
|
+
.map((img) => ({ image: { url: `data:${img.mime};base64,${img.data}` } }));
|
|
119
125
|
if (refs.length)
|
|
120
126
|
body.input_references = refs;
|
|
121
127
|
}
|
|
@@ -234,9 +240,7 @@ export async function handleGenerateVideo(request, apiClient, progress) {
|
|
|
234
240
|
if (!args.prompt || !args.prompt.trim()) {
|
|
235
241
|
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
236
242
|
}
|
|
237
|
-
const model = args.model ||
|
|
238
|
-
process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL ||
|
|
239
|
-
FALLBACK_MODEL;
|
|
243
|
+
const model = args.model || process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL || FALLBACK_MODEL;
|
|
240
244
|
// Audit entry — video is the most expensive tool we have. Always log
|
|
241
245
|
// model, resolution, duration, and a safe prompt preview so unintended
|
|
242
246
|
// spend can be traced.
|
|
@@ -20,7 +20,10 @@ export async function handleHealthCheck(_request, apiClient, modelCache) {
|
|
|
20
20
|
errorMessage = err instanceof Error ? err.message : String(err);
|
|
21
21
|
}
|
|
22
22
|
const modelsCached = modelCache.isValid() ? modelCache.size() : 0;
|
|
23
|
-
|
|
23
|
+
// `ok` means the API was reachable and the key was accepted. An empty
|
|
24
|
+
// catalog counts as success (the API just returned no models) — callers
|
|
25
|
+
// branch on `models_cached` if they care about the count.
|
|
26
|
+
const ok = apiKeyValid;
|
|
24
27
|
return buildStructuredResult({
|
|
25
28
|
ok,
|
|
26
29
|
server_version: SERVER_VERSION,
|
|
@@ -3,6 +3,7 @@ export declare const isBlockedIPv4: typeof _isBlockedIPv4;
|
|
|
3
3
|
export declare const assertUrlSafeForFetch: typeof _assertUrlSafeForFetch;
|
|
4
4
|
export declare function getMaxImageDimension(): number;
|
|
5
5
|
export declare function getImageJpegQuality(): number;
|
|
6
|
+
export declare function mimeFromExtension(ext: string): string | null;
|
|
6
7
|
export declare function getMimeType(filePath: string): string;
|
|
7
8
|
export declare function fetchHttpImage(urlString: string): Promise<Buffer>;
|
|
8
9
|
export declare function fetchImage(source: string): Promise<Buffer>;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import path from 'path';
|
|
2
2
|
import { promises as fs } from 'fs';
|
|
3
3
|
import { readEnvInt, isBlockedIPv4 as _isBlockedIPv4, assertUrlSafeForFetch as _assertUrlSafeForFetch, fetchHttpResource, parseBase64DataUrl, } from './fetch-utils.js';
|
|
4
|
+
import { resolveSafeInputPath } from './path-safety.js';
|
|
4
5
|
// Re-export for backward compatibility (tests import from image-utils)
|
|
5
6
|
export const isBlockedIPv4 = _isBlockedIPv4;
|
|
6
7
|
export const assertUrlSafeForFetch = _assertUrlSafeForFetch;
|
|
@@ -39,22 +40,29 @@ async function loadSharp() {
|
|
|
39
40
|
sharpFn = fn ?? mod;
|
|
40
41
|
}
|
|
41
42
|
catch {
|
|
42
|
-
|
|
43
|
+
// sharp is optional — images will be sent unprocessed (larger but functional)
|
|
44
|
+
const { logger } = await import('../logger.js');
|
|
45
|
+
logger.warn('sharp not available, images will be sent unprocessed');
|
|
43
46
|
}
|
|
44
47
|
}
|
|
45
48
|
return sharpFn;
|
|
46
49
|
}
|
|
50
|
+
const IMAGE_EXT_MIME = {
|
|
51
|
+
png: 'image/png',
|
|
52
|
+
jpg: 'image/jpeg',
|
|
53
|
+
jpeg: 'image/jpeg',
|
|
54
|
+
webp: 'image/webp',
|
|
55
|
+
gif: 'image/gif',
|
|
56
|
+
bmp: 'image/bmp',
|
|
57
|
+
};
|
|
58
|
+
export function mimeFromExtension(ext) {
|
|
59
|
+
const normalized = ext.toLowerCase().replace(/^\./, '');
|
|
60
|
+
if (!normalized)
|
|
61
|
+
return null;
|
|
62
|
+
return IMAGE_EXT_MIME[normalized] ?? null;
|
|
63
|
+
}
|
|
47
64
|
export function getMimeType(filePath) {
|
|
48
|
-
|
|
49
|
-
const map = {
|
|
50
|
-
'.png': 'image/png',
|
|
51
|
-
'.jpg': 'image/jpeg',
|
|
52
|
-
'.jpeg': 'image/jpeg',
|
|
53
|
-
'.webp': 'image/webp',
|
|
54
|
-
'.gif': 'image/gif',
|
|
55
|
-
'.bmp': 'image/bmp',
|
|
56
|
-
};
|
|
57
|
-
return map[ext] || 'image/jpeg';
|
|
65
|
+
return mimeFromExtension(path.extname(filePath)) ?? 'image/jpeg';
|
|
58
66
|
}
|
|
59
67
|
export async function fetchHttpImage(urlString) {
|
|
60
68
|
const { buffer } = await fetchHttpResource(urlString, {
|
|
@@ -77,7 +85,8 @@ export async function fetchImage(source) {
|
|
|
77
85
|
if (source.startsWith('http://') || source.startsWith('https://')) {
|
|
78
86
|
return fetchHttpImage(source);
|
|
79
87
|
}
|
|
80
|
-
|
|
88
|
+
const safe = await resolveSafeInputPath(source);
|
|
89
|
+
return fs.readFile(safe);
|
|
81
90
|
}
|
|
82
91
|
/**
|
|
83
92
|
* Sniff image MIME type from magic bytes. Used to label the output of a
|
|
@@ -134,13 +143,14 @@ export async function optimizeImage(buffer) {
|
|
|
134
143
|
const maxDim = getMaxImageDimension();
|
|
135
144
|
const quality = getImageJpegQuality();
|
|
136
145
|
try {
|
|
137
|
-
const
|
|
138
|
-
|
|
146
|
+
const pipeline = sharp(buffer);
|
|
147
|
+
const meta = await pipeline.metadata();
|
|
148
|
+
let resized = pipeline;
|
|
139
149
|
if (meta.width && meta.height && Math.max(meta.width, meta.height) > maxDim) {
|
|
140
150
|
const opts = meta.width > meta.height ? { width: maxDim } : { height: maxDim };
|
|
141
|
-
|
|
151
|
+
resized = pipeline.resize(opts);
|
|
142
152
|
}
|
|
143
|
-
const out = await
|
|
153
|
+
const out = await resized.jpeg({ quality }).toBuffer();
|
|
144
154
|
return { base64: out.toString('base64'), mime: 'image/jpeg' };
|
|
145
155
|
}
|
|
146
156
|
catch {
|
|
@@ -14,5 +14,9 @@ import { type ToolErrorResult } from '../errors.js';
|
|
|
14
14
|
* 2. Message heuristics for common OpenRouter strings (credits, ZDR,
|
|
15
15
|
* "model does not exist", content policy, etc.).
|
|
16
16
|
* 3. Default to INTERNAL to avoid leaking raw shapes.
|
|
17
|
+
*
|
|
18
|
+
* When the error carries a `Retry-After` header (on 429 / 503) we populate
|
|
19
|
+
* `_meta.retry_after_seconds` so agents can back off intelligently. We
|
|
20
|
+
* also attach canonical `suggestions[]` for common cases.
|
|
17
21
|
*/
|
|
18
|
-
export declare function classifyUpstreamError(err: unknown,
|
|
22
|
+
export declare function classifyUpstreamError(err: unknown, contextMessage?: string): ToolErrorResult;
|