@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +367 -283
- package/dist/index.js +1 -1
- package/dist/model-cache.d.ts +22 -12
- package/dist/model-cache.js +58 -21
- package/dist/openrouter-api.d.ts +45 -0
- package/dist/openrouter-api.js +50 -0
- package/dist/tool-descriptions.d.ts +19 -0
- package/dist/tool-descriptions.js +573 -0
- package/dist/tool-handlers/analyze-audio.js +5 -1
- package/dist/tool-handlers/analyze-image.js +6 -5
- package/dist/tool-handlers/analyze-video.js +6 -5
- package/dist/tool-handlers/async-chat.d.ts +51 -0
- package/dist/tool-handlers/async-chat.js +216 -0
- package/dist/tool-handlers/audio-utils.js +4 -2
- package/dist/tool-handlers/chat-completion.js +1 -1
- package/dist/tool-handlers/fetch-utils.js +16 -2
- package/dist/tool-handlers/generate-audio.js +2 -4
- package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
- package/dist/tool-handlers/generate-image-dedicated.js +176 -0
- package/dist/tool-handlers/generate-image-input.d.ts +3 -0
- package/dist/tool-handlers/generate-image-input.js +38 -0
- package/dist/tool-handlers/generate-image.d.ts +13 -51
- package/dist/tool-handlers/generate-image.js +32 -119
- package/dist/tool-handlers/generate-video.d.ts +2 -2
- package/dist/tool-handlers/generate-video.js +78 -30
- package/dist/tool-handlers/image-utils.d.ts +1 -0
- package/dist/tool-handlers/image-utils.js +26 -16
- package/dist/tool-handlers/openrouter-errors.js +6 -2
- package/dist/tool-handlers/path-safety.js +32 -5
- package/dist/tool-handlers/provider-routing.js +7 -2
- package/dist/tool-handlers/rerank.js +2 -5
- package/dist/tool-handlers/search-models.d.ts +2 -2
- package/dist/tool-handlers/search-models.js +2 -6
- package/dist/tool-handlers/speech-to-text.d.ts +20 -0
- package/dist/tool-handlers/speech-to-text.js +140 -0
- package/dist/tool-handlers/structured-output.d.ts +8 -0
- package/dist/tool-handlers/structured-output.js +11 -0
- package/dist/tool-handlers/text-to-speech.d.ts +29 -0
- package/dist/tool-handlers/text-to-speech.js +105 -0
- package/dist/tool-handlers/video-utils.js +6 -9
- package/dist/tool-handlers.js +253 -125
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +27 -15
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import OpenAI from 'openai';
|
|
2
|
+
import type { ChatCompletionMessageParam } from 'openai/resources/chat/completions.js';
|
|
3
|
+
import { type ProviderRoutingOptions } from './provider-routing.js';
|
|
4
|
+
import { type CacheOptions } from './cache.js';
|
|
5
|
+
export interface StartChatCompletionRequest extends CacheOptions {
|
|
6
|
+
messages: ChatCompletionMessageParam[];
|
|
7
|
+
model?: string;
|
|
8
|
+
temperature?: number;
|
|
9
|
+
max_tokens?: number;
|
|
10
|
+
provider?: ProviderRoutingOptions;
|
|
11
|
+
include_reasoning?: boolean;
|
|
12
|
+
online?: boolean;
|
|
13
|
+
web_max_results?: number;
|
|
14
|
+
}
|
|
15
|
+
export interface GetChatCompletionStatusRequest {
|
|
16
|
+
job_id: string;
|
|
17
|
+
}
|
|
18
|
+
export type AsyncJobStatus = 'queued' | 'running' | 'completed' | 'failed';
|
|
19
|
+
export declare function handleStartChatCompletion(request: {
|
|
20
|
+
params: {
|
|
21
|
+
arguments: StartChatCompletionRequest;
|
|
22
|
+
};
|
|
23
|
+
}, openai: OpenAI, defaultModel?: string): Promise<import("../errors.js").ToolErrorResult | {
|
|
24
|
+
content: {
|
|
25
|
+
type: "text";
|
|
26
|
+
text: string;
|
|
27
|
+
}[];
|
|
28
|
+
_meta: {
|
|
29
|
+
server_version: string;
|
|
30
|
+
job_id: string;
|
|
31
|
+
status: "running";
|
|
32
|
+
model: string;
|
|
33
|
+
};
|
|
34
|
+
}>;
|
|
35
|
+
export declare function handleGetChatCompletionStatus(request: {
|
|
36
|
+
params: {
|
|
37
|
+
arguments: GetChatCompletionStatusRequest;
|
|
38
|
+
};
|
|
39
|
+
}): Promise<import("../errors.js").ToolErrorResult | {
|
|
40
|
+
content: {
|
|
41
|
+
type: "text";
|
|
42
|
+
text: string;
|
|
43
|
+
}[];
|
|
44
|
+
_meta: {
|
|
45
|
+
server_version: string;
|
|
46
|
+
job_id: string;
|
|
47
|
+
status: "queued" | "completed" | "running";
|
|
48
|
+
model: string;
|
|
49
|
+
created_at: string;
|
|
50
|
+
};
|
|
51
|
+
}>;
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Async chat completions — resumable workflow for long-running requests.
|
|
3
|
+
*
|
|
4
|
+
* Problem: Remote MCP bridges (Cowork, etc.) kill tool calls after ~60s.
|
|
5
|
+
* Reasoning models can take much longer. Unlike video, `chat_completion`
|
|
6
|
+
* currently has no background job mechanism.
|
|
7
|
+
*
|
|
8
|
+
* Solution: Two tools that mirror the video pattern:
|
|
9
|
+
* - `start_chat_completion` — fires off the request in the background,
|
|
10
|
+
* returns a `job_id` immediately.
|
|
11
|
+
* - `get_chat_completion_status` — returns queued/running/completed/failed,
|
|
12
|
+
* with the final response on completion.
|
|
13
|
+
*
|
|
14
|
+
* Job state is held in memory (survives within a single MCP session).
|
|
15
|
+
* Optionally persisted to OPENROUTER_OUTPUT_DIR/openrouter-jobs/ for
|
|
16
|
+
* crash recovery.
|
|
17
|
+
*/
|
|
18
|
+
import { promises as fs } from 'fs';
|
|
19
|
+
import path from 'node:path';
|
|
20
|
+
import { ErrorCode, toolError } from '../errors.js';
|
|
21
|
+
import { SERVER_VERSION } from '../version.js';
|
|
22
|
+
import { logger } from '../logger.js';
|
|
23
|
+
import { extractCompletionText, buildCompletionMeta, } from './completion-utils.js';
|
|
24
|
+
import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
|
|
25
|
+
import { buildCacheHeaders } from './cache.js';
|
|
26
|
+
// ─── Job Store ───────────────────────────────────────────────────────────────
|
|
27
|
+
const jobs = new Map();
|
|
28
|
+
let jobCounter = 0;
|
|
29
|
+
function generateJobId() {
|
|
30
|
+
jobCounter += 1;
|
|
31
|
+
const ts = new Date().toISOString().replace(/[-:T]/g, '').slice(0, 14);
|
|
32
|
+
return `chat_${ts}_${String(jobCounter).padStart(3, '0')}`;
|
|
33
|
+
}
|
|
34
|
+
function getJobsDir() {
|
|
35
|
+
const outputDir = process.env.OPENROUTER_OUTPUT_DIR;
|
|
36
|
+
if (!outputDir)
|
|
37
|
+
return null;
|
|
38
|
+
return path.join(outputDir, 'openrouter-jobs');
|
|
39
|
+
}
|
|
40
|
+
async function persistJob(job) {
|
|
41
|
+
const dir = getJobsDir();
|
|
42
|
+
if (!dir)
|
|
43
|
+
return;
|
|
44
|
+
try {
|
|
45
|
+
const jobDir = path.join(dir, job.id);
|
|
46
|
+
await fs.mkdir(jobDir, { recursive: true });
|
|
47
|
+
await fs.writeFile(path.join(jobDir, 'status.json'), JSON.stringify(job, null, 2));
|
|
48
|
+
if (job.status === 'completed' && job.result?.text) {
|
|
49
|
+
await fs.writeFile(path.join(jobDir, 'response.md'), job.result.text);
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
catch (err) {
|
|
53
|
+
logger.warn('async_chat.persist_error', {
|
|
54
|
+
job_id: job.id,
|
|
55
|
+
err: err instanceof Error ? err.message : String(err),
|
|
56
|
+
});
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
// ─── Handlers ────────────────────────────────────────────────────────────────
|
|
60
|
+
const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
|
|
61
|
+
function readIncludeReasoningDefault() {
|
|
62
|
+
const raw = (process.env.OPENROUTER_INCLUDE_REASONING ?? '').trim().toLowerCase();
|
|
63
|
+
return raw === '1' || raw === 'true' || raw === 'yes';
|
|
64
|
+
}
|
|
65
|
+
export async function handleStartChatCompletion(request, openai, defaultModel) {
|
|
66
|
+
const args = request.params.arguments ?? { messages: [] };
|
|
67
|
+
const { messages, model, temperature, max_tokens, provider, include_reasoning, online, web_max_results, cache, cache_ttl, cache_clear, } = args;
|
|
68
|
+
if (!messages?.length) {
|
|
69
|
+
return toolError(ErrorCode.INVALID_INPUT, 'Messages array cannot be empty.');
|
|
70
|
+
}
|
|
71
|
+
const effectiveModel = model || defaultModel || DEFAULT_MODEL;
|
|
72
|
+
const jobId = generateJobId();
|
|
73
|
+
// Create the job immediately
|
|
74
|
+
const job = {
|
|
75
|
+
id: jobId,
|
|
76
|
+
status: 'running',
|
|
77
|
+
createdAt: new Date().toISOString(),
|
|
78
|
+
model: effectiveModel,
|
|
79
|
+
};
|
|
80
|
+
jobs.set(jobId, job);
|
|
81
|
+
logger.audit('async_chat.start', {
|
|
82
|
+
job_id: jobId,
|
|
83
|
+
model: effectiveModel,
|
|
84
|
+
message_count: messages.length,
|
|
85
|
+
});
|
|
86
|
+
// Fire and forget — the completion runs in the background
|
|
87
|
+
runCompletionInBackground(job, openai, {
|
|
88
|
+
messages,
|
|
89
|
+
model: effectiveModel,
|
|
90
|
+
temperature,
|
|
91
|
+
max_tokens,
|
|
92
|
+
provider,
|
|
93
|
+
include_reasoning,
|
|
94
|
+
online,
|
|
95
|
+
web_max_results,
|
|
96
|
+
cache,
|
|
97
|
+
cache_ttl,
|
|
98
|
+
cache_clear,
|
|
99
|
+
});
|
|
100
|
+
// Return immediately with the job ID
|
|
101
|
+
return {
|
|
102
|
+
content: [
|
|
103
|
+
{
|
|
104
|
+
type: 'text',
|
|
105
|
+
text: `Chat completion job started. Use get_chat_completion_status with job_id="${jobId}" to check results.`,
|
|
106
|
+
},
|
|
107
|
+
],
|
|
108
|
+
_meta: {
|
|
109
|
+
server_version: SERVER_VERSION,
|
|
110
|
+
job_id: jobId,
|
|
111
|
+
status: 'running',
|
|
112
|
+
model: effectiveModel,
|
|
113
|
+
},
|
|
114
|
+
};
|
|
115
|
+
}
|
|
116
|
+
async function runCompletionInBackground(job, openai, opts) {
|
|
117
|
+
const providerOptions = mergeProviderOptions(readProviderDefaults(), opts.provider);
|
|
118
|
+
const providerBody = buildProviderBody(providerOptions);
|
|
119
|
+
const effectiveMaxTokens = resolveMaxTokens(opts.max_tokens);
|
|
120
|
+
const wantsReasoning = opts.include_reasoning ?? readIncludeReasoningDefault();
|
|
121
|
+
const body = {
|
|
122
|
+
model: opts.model,
|
|
123
|
+
messages: opts.messages,
|
|
124
|
+
temperature: opts.temperature ?? 1,
|
|
125
|
+
};
|
|
126
|
+
if (typeof effectiveMaxTokens === 'number')
|
|
127
|
+
body.max_tokens = effectiveMaxTokens;
|
|
128
|
+
if (providerBody)
|
|
129
|
+
body.provider = providerBody;
|
|
130
|
+
if (wantsReasoning)
|
|
131
|
+
body.include_reasoning = true;
|
|
132
|
+
if (opts.online) {
|
|
133
|
+
const plugin = { id: 'web' };
|
|
134
|
+
if (typeof opts.web_max_results === 'number' && opts.web_max_results > 0) {
|
|
135
|
+
plugin.max_results = opts.web_max_results;
|
|
136
|
+
}
|
|
137
|
+
body.plugins = [plugin];
|
|
138
|
+
}
|
|
139
|
+
const headers = buildCacheHeaders({
|
|
140
|
+
cache: opts.cache,
|
|
141
|
+
cache_ttl: opts.cache_ttl,
|
|
142
|
+
cache_clear: opts.cache_clear,
|
|
143
|
+
});
|
|
144
|
+
const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
|
|
145
|
+
try {
|
|
146
|
+
const completion = (await openai.chat.completions.create(body, requestOpts));
|
|
147
|
+
const extracted = extractCompletionText(completion);
|
|
148
|
+
if (!extracted.text) {
|
|
149
|
+
job.status = 'failed';
|
|
150
|
+
job.error = 'Model returned no textual content.';
|
|
151
|
+
}
|
|
152
|
+
else {
|
|
153
|
+
job.status = 'completed';
|
|
154
|
+
job.result = {
|
|
155
|
+
text: extracted.text,
|
|
156
|
+
meta: buildCompletionMeta(extracted, {
|
|
157
|
+
includeReasoning: wantsReasoning,
|
|
158
|
+
extra: { server_version: SERVER_VERSION },
|
|
159
|
+
}),
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
catch (err) {
|
|
164
|
+
job.status = 'failed';
|
|
165
|
+
job.error = err instanceof Error ? err.message : String(err);
|
|
166
|
+
logger.warn('async_chat.failed', { job_id: job.id, error: job.error });
|
|
167
|
+
}
|
|
168
|
+
await persistJob(job);
|
|
169
|
+
}
|
|
170
|
+
export async function handleGetChatCompletionStatus(request) {
|
|
171
|
+
const args = request.params.arguments ?? {};
|
|
172
|
+
const jobId = args.job_id?.trim();
|
|
173
|
+
if (!jobId) {
|
|
174
|
+
return toolError(ErrorCode.INVALID_INPUT, 'job_id is required.');
|
|
175
|
+
}
|
|
176
|
+
const job = jobs.get(jobId);
|
|
177
|
+
if (!job) {
|
|
178
|
+
return toolError(ErrorCode.INVALID_INPUT, `No job found with id "${jobId}". Jobs are stored in memory for the current session only.`);
|
|
179
|
+
}
|
|
180
|
+
if (job.status === 'completed' && job.result) {
|
|
181
|
+
return {
|
|
182
|
+
content: [{ type: 'text', text: job.result.text }],
|
|
183
|
+
_meta: {
|
|
184
|
+
server_version: SERVER_VERSION,
|
|
185
|
+
job_id: jobId,
|
|
186
|
+
status: 'completed',
|
|
187
|
+
model: job.model,
|
|
188
|
+
created_at: job.createdAt,
|
|
189
|
+
...job.result.meta,
|
|
190
|
+
},
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
if (job.status === 'failed') {
|
|
194
|
+
return toolError(ErrorCode.JOB_FAILED, job.error || 'Job failed.', {
|
|
195
|
+
job_id: jobId,
|
|
196
|
+
model: job.model,
|
|
197
|
+
created_at: job.createdAt,
|
|
198
|
+
});
|
|
199
|
+
}
|
|
200
|
+
// Still running
|
|
201
|
+
return {
|
|
202
|
+
content: [
|
|
203
|
+
{
|
|
204
|
+
type: 'text',
|
|
205
|
+
text: `Job ${jobId} is still ${job.status}. Try again in a few seconds.`,
|
|
206
|
+
},
|
|
207
|
+
],
|
|
208
|
+
_meta: {
|
|
209
|
+
server_version: SERVER_VERSION,
|
|
210
|
+
job_id: jobId,
|
|
211
|
+
status: job.status,
|
|
212
|
+
model: job.model,
|
|
213
|
+
created_at: job.createdAt,
|
|
214
|
+
},
|
|
215
|
+
};
|
|
216
|
+
}
|
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
import path from 'path';
|
|
6
6
|
import { promises as fs } from 'fs';
|
|
7
7
|
import { readEnvInt, fetchHttpResource, parseBase64DataUrl } from './fetch-utils.js';
|
|
8
|
+
import { resolveSafeInputPath } from './path-safety.js';
|
|
8
9
|
// Re-export for tests
|
|
9
10
|
export { isBlockedIPv4, assertUrlSafeForFetch } from './fetch-utils.js';
|
|
10
11
|
const DEFAULT_FETCH_TIMEOUT_MS = 30_000;
|
|
@@ -117,10 +118,11 @@ export async function prepareAudioData(source) {
|
|
|
117
118
|
return { data: buffer.toString('base64'), format };
|
|
118
119
|
}
|
|
119
120
|
// --- local file ---
|
|
120
|
-
const
|
|
121
|
+
const safe = await resolveSafeInputPath(source);
|
|
122
|
+
const format = getAudioFormat(safe);
|
|
121
123
|
if (!format) {
|
|
122
124
|
throw new Error(`Unsupported audio format for file: ${source}. Supported: ${SUPPORTED_AUDIO_FORMATS.join(', ')}`);
|
|
123
125
|
}
|
|
124
|
-
const buffer = await fs.readFile(
|
|
126
|
+
const buffer = await fs.readFile(safe);
|
|
125
127
|
return { data: buffer.toString('base64'), format };
|
|
126
128
|
}
|
|
@@ -3,7 +3,7 @@ import { SERVER_VERSION } from '../version.js';
|
|
|
3
3
|
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
4
4
|
import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
|
|
5
5
|
import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
|
|
6
|
-
import { buildCacheHeaders, extractCacheMeta
|
|
6
|
+
import { buildCacheHeaders, extractCacheMeta } from './cache.js';
|
|
7
7
|
import { awaitCompletionWithHeaders } from './openai-withresponse.js';
|
|
8
8
|
const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
|
|
9
9
|
function readIncludeReasoningDefault() {
|
|
@@ -151,11 +151,25 @@ export function isBlockedIPv6(ip) {
|
|
|
151
151
|
return false;
|
|
152
152
|
const [g0, g1, g2, g3, g4, g5, g6, g7] = groups;
|
|
153
153
|
// :: (unspecified)
|
|
154
|
-
if (g0 === 0 &&
|
|
154
|
+
if (g0 === 0 &&
|
|
155
|
+
g1 === 0 &&
|
|
156
|
+
g2 === 0 &&
|
|
157
|
+
g3 === 0 &&
|
|
158
|
+
g4 === 0 &&
|
|
159
|
+
g5 === 0 &&
|
|
160
|
+
g6 === 0 &&
|
|
161
|
+
g7 === 0) {
|
|
155
162
|
return true;
|
|
156
163
|
}
|
|
157
164
|
// ::1 (loopback)
|
|
158
|
-
if (g0 === 0 &&
|
|
165
|
+
if (g0 === 0 &&
|
|
166
|
+
g1 === 0 &&
|
|
167
|
+
g2 === 0 &&
|
|
168
|
+
g3 === 0 &&
|
|
169
|
+
g4 === 0 &&
|
|
170
|
+
g5 === 0 &&
|
|
171
|
+
g6 === 0 &&
|
|
172
|
+
g7 === 1) {
|
|
159
173
|
return true;
|
|
160
174
|
}
|
|
161
175
|
// ::ffff:0:0/96 — IPv4-mapped. Re-check the embedded IPv4.
|
|
@@ -95,9 +95,7 @@ export async function handleGenerateAudio(request, openai) {
|
|
|
95
95
|
logger.audit('generate_audio.start', {
|
|
96
96
|
model: model || DEFAULT_MODEL,
|
|
97
97
|
voice: voice?.trim() || DEFAULT_VOICE,
|
|
98
|
-
format: VALID_FORMATS.includes(format ?? '')
|
|
99
|
-
? format
|
|
100
|
-
: DEFAULT_FORMAT,
|
|
98
|
+
format: VALID_FORMATS.includes(format ?? '') ? format : DEFAULT_FORMAT,
|
|
101
99
|
prompt_preview: prompt.slice(0, 80),
|
|
102
100
|
save_path: save_path ? 'provided' : 'none',
|
|
103
101
|
});
|
|
@@ -156,7 +154,7 @@ export async function handleGenerateAudio(request, openai) {
|
|
|
156
154
|
const detected = detectAudioFormat(audioBuffer);
|
|
157
155
|
// Always wrap raw PCM in WAV so it's playable
|
|
158
156
|
if (detected.ext === 'pcm') {
|
|
159
|
-
audioBuffer = wrapPcmInWav(audioBuffer);
|
|
157
|
+
audioBuffer = Buffer.from(wrapPcmInWav(audioBuffer));
|
|
160
158
|
detected.ext = 'wav';
|
|
161
159
|
detected.mimeType = 'audio/wav';
|
|
162
160
|
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import type { OpenRouterAPIClient } from '../openrouter-api.js';
|
|
2
|
+
import { type CacheOptions } from './cache.js';
|
|
3
|
+
export interface GenerateImageDedicatedRequest extends CacheOptions {
|
|
4
|
+
prompt: string;
|
|
5
|
+
model?: string;
|
|
6
|
+
resolution?: string;
|
|
7
|
+
aspect_ratio?: string;
|
|
8
|
+
quality?: string;
|
|
9
|
+
output_format?: string;
|
|
10
|
+
n?: number;
|
|
11
|
+
input_references?: string[];
|
|
12
|
+
save_path?: string;
|
|
13
|
+
provider?: Record<string, unknown>;
|
|
14
|
+
}
|
|
15
|
+
export declare function handleGenerateImageDedicated(request: {
|
|
16
|
+
params: {
|
|
17
|
+
arguments: GenerateImageDedicatedRequest;
|
|
18
|
+
};
|
|
19
|
+
}, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
|
|
20
|
+
content: ({
|
|
21
|
+
type: "text";
|
|
22
|
+
text: string;
|
|
23
|
+
mimeType?: undefined;
|
|
24
|
+
data?: undefined;
|
|
25
|
+
} | {
|
|
26
|
+
type: "image";
|
|
27
|
+
mimeType: string;
|
|
28
|
+
data: string;
|
|
29
|
+
text?: undefined;
|
|
30
|
+
})[];
|
|
31
|
+
_meta: Record<string, unknown>;
|
|
32
|
+
}>;
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* generate_image_dedicated — uses OpenRouter's dedicated POST /api/v1/images
|
|
3
|
+
* endpoint (launched June 2026) for image generation. Supports normalized
|
|
4
|
+
* resolution tiers, aspect ratios, quality levels, output formats, and
|
|
5
|
+
* input_references for image-to-image workflows.
|
|
6
|
+
*
|
|
7
|
+
* This is distinct from the original `generate_image` tool which uses chat
|
|
8
|
+
* completions with `modalities: ['image', 'text']`. New image models are
|
|
9
|
+
* added exclusively to this dedicated endpoint.
|
|
10
|
+
*/
|
|
11
|
+
import { promises as fs } from 'fs';
|
|
12
|
+
import path from 'node:path';
|
|
13
|
+
import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
14
|
+
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
15
|
+
import { SERVER_VERSION } from '../version.js';
|
|
16
|
+
import { logger } from '../logger.js';
|
|
17
|
+
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
18
|
+
import { buildCacheHeaders } from './cache.js';
|
|
19
|
+
const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
|
|
20
|
+
const VALID_RESOLUTIONS = new Set(['512', '0.5K', '1K', '2K', '4K']);
|
|
21
|
+
const VALID_QUALITIES = new Set(['auto', 'low', 'medium', 'high']);
|
|
22
|
+
const VALID_OUTPUT_FORMATS = new Set(['png', 'jpeg', 'webp', 'svg']);
|
|
23
|
+
/**
|
|
24
|
+
* Resolve an input image reference (local path, URL, or data URL) into the
|
|
25
|
+
* OpenRouter `input_references` shape: `{ type: "image_url", image_url: { url } }`.
|
|
26
|
+
*/
|
|
27
|
+
async function resolveReference(source) {
|
|
28
|
+
const trimmed = source.trim();
|
|
29
|
+
if (!trimmed)
|
|
30
|
+
throw new Error('Empty input_references entry');
|
|
31
|
+
// Data URLs and HTTP URLs pass through directly
|
|
32
|
+
if (trimmed.startsWith('data:') || /^https?:\/\//i.test(trimmed)) {
|
|
33
|
+
return { type: 'image_url', image_url: { url: trimmed } };
|
|
34
|
+
}
|
|
35
|
+
// Local file: sandbox, read, and convert to data URL
|
|
36
|
+
const abs = await resolveSafeInputPath(trimmed);
|
|
37
|
+
const buf = await fs.readFile(abs);
|
|
38
|
+
const ext = path.extname(abs).toLowerCase();
|
|
39
|
+
const mime = ext === '.png' ? 'image/png' :
|
|
40
|
+
ext === '.webp' ? 'image/webp' :
|
|
41
|
+
ext === '.gif' ? 'image/gif' :
|
|
42
|
+
ext === '.svg' ? 'image/svg+xml' :
|
|
43
|
+
'image/jpeg';
|
|
44
|
+
const dataUrl = `data:${mime};base64,${buf.toString('base64')}`;
|
|
45
|
+
return { type: 'image_url', image_url: { url: dataUrl } };
|
|
46
|
+
}
|
|
47
|
+
export async function handleGenerateImageDedicated(request, apiClient) {
|
|
48
|
+
const args = request.params.arguments ?? {};
|
|
49
|
+
const { prompt, model, resolution, aspect_ratio, quality, output_format, n, input_references, save_path, provider, cache, cache_ttl, cache_clear, } = args;
|
|
50
|
+
if (!prompt?.trim()) {
|
|
51
|
+
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
52
|
+
}
|
|
53
|
+
logger.audit('generate_image_dedicated.start', {
|
|
54
|
+
model: model || DEFAULT_MODEL,
|
|
55
|
+
prompt_preview: prompt.slice(0, 80),
|
|
56
|
+
resolution,
|
|
57
|
+
aspect_ratio,
|
|
58
|
+
quality,
|
|
59
|
+
output_format,
|
|
60
|
+
input_references_count: input_references?.length ?? 0,
|
|
61
|
+
save_path: save_path ? 'provided' : 'none',
|
|
62
|
+
});
|
|
63
|
+
// Validate enums
|
|
64
|
+
if (resolution && !VALID_RESOLUTIONS.has(resolution)) {
|
|
65
|
+
return toolError(ErrorCode.INVALID_INPUT, `resolution '${resolution}' is not supported. Valid: ${[...VALID_RESOLUTIONS].join(', ')}.`);
|
|
66
|
+
}
|
|
67
|
+
if (quality && !VALID_QUALITIES.has(quality)) {
|
|
68
|
+
return toolError(ErrorCode.INVALID_INPUT, `quality '${quality}' is not supported. Valid: ${[...VALID_QUALITIES].join(', ')}.`);
|
|
69
|
+
}
|
|
70
|
+
if (output_format && !VALID_OUTPUT_FORMATS.has(output_format)) {
|
|
71
|
+
return toolError(ErrorCode.INVALID_INPUT, `output_format '${output_format}' is not supported. Valid: ${[...VALID_OUTPUT_FORMATS].join(', ')}.`);
|
|
72
|
+
}
|
|
73
|
+
// Resolve save path early
|
|
74
|
+
let safeSavePath = null;
|
|
75
|
+
if (save_path) {
|
|
76
|
+
try {
|
|
77
|
+
safeSavePath = await resolveSafeOutputPath(save_path);
|
|
78
|
+
}
|
|
79
|
+
catch (err) {
|
|
80
|
+
if (err instanceof UnsafeOutputPathError)
|
|
81
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
82
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
// Build request body
|
|
86
|
+
const body = {
|
|
87
|
+
model: model || DEFAULT_MODEL,
|
|
88
|
+
prompt,
|
|
89
|
+
};
|
|
90
|
+
if (resolution)
|
|
91
|
+
body.resolution = resolution;
|
|
92
|
+
if (aspect_ratio)
|
|
93
|
+
body.aspect_ratio = aspect_ratio;
|
|
94
|
+
if (quality)
|
|
95
|
+
body.quality = quality;
|
|
96
|
+
if (output_format)
|
|
97
|
+
body.output_format = output_format;
|
|
98
|
+
if (typeof n === 'number' && n > 0)
|
|
99
|
+
body.n = n;
|
|
100
|
+
if (provider && typeof provider === 'object')
|
|
101
|
+
body.provider = provider;
|
|
102
|
+
// Resolve input references
|
|
103
|
+
if (input_references?.length) {
|
|
104
|
+
try {
|
|
105
|
+
const refs = await Promise.all(input_references.map(resolveReference));
|
|
106
|
+
body.input_references = refs;
|
|
107
|
+
}
|
|
108
|
+
catch (err) {
|
|
109
|
+
if (err instanceof UnsafeOutputPathError)
|
|
110
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
111
|
+
return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_references');
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
// Build cache headers
|
|
115
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
116
|
+
let response;
|
|
117
|
+
try {
|
|
118
|
+
response = await apiClient.generateImage(body, headers);
|
|
119
|
+
}
|
|
120
|
+
catch (err) {
|
|
121
|
+
return classifyUpstreamError(err, 'generate_image_dedicated');
|
|
122
|
+
}
|
|
123
|
+
const images = response.data ?? [];
|
|
124
|
+
if (!images.length || (!images[0]?.b64_json && !images[0]?.url)) {
|
|
125
|
+
return toolError(ErrorCode.UPSTREAM_REFUSED, 'Model returned no image data.', {
|
|
126
|
+
response_keys: Object.keys(response),
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
const firstImage = images[0];
|
|
130
|
+
const imageData = firstImage.b64_json;
|
|
131
|
+
const mimeType = output_format === 'png' ? 'image/png' :
|
|
132
|
+
output_format === 'webp' ? 'image/webp' :
|
|
133
|
+
output_format === 'svg' ? 'image/svg+xml' :
|
|
134
|
+
output_format === 'jpeg' ? 'image/jpeg' :
|
|
135
|
+
'image/png'; // default
|
|
136
|
+
const baseMeta = {
|
|
137
|
+
server_version: SERVER_VERSION,
|
|
138
|
+
model: model || DEFAULT_MODEL,
|
|
139
|
+
images_count: images.length,
|
|
140
|
+
};
|
|
141
|
+
if (response.usage)
|
|
142
|
+
baseMeta.usage = response.usage;
|
|
143
|
+
if (firstImage.revised_prompt)
|
|
144
|
+
baseMeta.revised_prompt = firstImage.revised_prompt;
|
|
145
|
+
// Save to file if requested
|
|
146
|
+
if (safeSavePath && imageData) {
|
|
147
|
+
try {
|
|
148
|
+
await fs.writeFile(safeSavePath, imageData, { encoding: 'base64' });
|
|
149
|
+
}
|
|
150
|
+
catch (err) {
|
|
151
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
|
|
152
|
+
}
|
|
153
|
+
baseMeta.save_path = safeSavePath;
|
|
154
|
+
return {
|
|
155
|
+
content: [
|
|
156
|
+
{ type: 'text', text: `Image saved to: ${safeSavePath}` },
|
|
157
|
+
...(imageData ? [{ type: 'image', mimeType, data: imageData }] : []),
|
|
158
|
+
],
|
|
159
|
+
_meta: baseMeta,
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
// Return inline
|
|
163
|
+
if (imageData) {
|
|
164
|
+
return {
|
|
165
|
+
content: [{ type: 'image', mimeType, data: imageData }],
|
|
166
|
+
_meta: baseMeta,
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
// URL-only response (some models return URLs instead of base64)
|
|
170
|
+
return {
|
|
171
|
+
content: [
|
|
172
|
+
{ type: 'text', text: `Image generated. URL: ${firstImage.url}` },
|
|
173
|
+
],
|
|
174
|
+
_meta: { ...baseMeta, image_url: firstImage.url },
|
|
175
|
+
};
|
|
176
|
+
}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import { promises as fs } from 'fs';
|
|
2
|
+
import path from 'node:path';
|
|
3
|
+
import { resolveSafeInputPath } from './path-safety.js';
|
|
4
|
+
import { mimeFromExtension } from './image-utils.js';
|
|
5
|
+
export async function resolveInputImage(ref) {
|
|
6
|
+
const trimmed = ref.trim();
|
|
7
|
+
if (!trimmed)
|
|
8
|
+
throw new Error('empty input_images entry');
|
|
9
|
+
if (trimmed.startsWith('data:'))
|
|
10
|
+
return trimmed;
|
|
11
|
+
if (/^https?:\/\//i.test(trimmed))
|
|
12
|
+
return trimmed;
|
|
13
|
+
const abs = await resolveSafeInputPath(trimmed);
|
|
14
|
+
const buf = await fs.readFile(abs);
|
|
15
|
+
const mime = mimeFromExtension(path.extname(abs)) || 'image/png';
|
|
16
|
+
return `data:${mime};base64,${buf.toString('base64')}`;
|
|
17
|
+
}
|
|
18
|
+
export async function buildUserContent(prompt, inputImages) {
|
|
19
|
+
if (!inputImages?.length) {
|
|
20
|
+
return `Generate an image: ${prompt}`;
|
|
21
|
+
}
|
|
22
|
+
const parts = [
|
|
23
|
+
{
|
|
24
|
+
type: 'text',
|
|
25
|
+
text: `Generate an image based on this prompt, using the following reference image(s) ` +
|
|
26
|
+
`for visual consistency. Match the appearance, identity, and style of the references ` +
|
|
27
|
+
`closely; do not alter them.\n\nPrompt: ${prompt}`,
|
|
28
|
+
},
|
|
29
|
+
];
|
|
30
|
+
const urls = await Promise.all(inputImages.map((ref) => resolveInputImage(ref)));
|
|
31
|
+
for (const url of urls) {
|
|
32
|
+
parts.push({
|
|
33
|
+
type: 'image_url',
|
|
34
|
+
image_url: { url, detail: 'high' },
|
|
35
|
+
});
|
|
36
|
+
}
|
|
37
|
+
return parts;
|
|
38
|
+
}
|