@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +367 -283
  2. package/dist/index.js +1 -1
  3. package/dist/model-cache.d.ts +22 -12
  4. package/dist/model-cache.js +58 -21
  5. package/dist/openrouter-api.d.ts +45 -0
  6. package/dist/openrouter-api.js +50 -0
  7. package/dist/tool-descriptions.d.ts +19 -0
  8. package/dist/tool-descriptions.js +573 -0
  9. package/dist/tool-handlers/analyze-audio.js +5 -1
  10. package/dist/tool-handlers/analyze-image.js +6 -5
  11. package/dist/tool-handlers/analyze-video.js +6 -5
  12. package/dist/tool-handlers/async-chat.d.ts +51 -0
  13. package/dist/tool-handlers/async-chat.js +216 -0
  14. package/dist/tool-handlers/audio-utils.js +4 -2
  15. package/dist/tool-handlers/chat-completion.js +1 -1
  16. package/dist/tool-handlers/fetch-utils.js +16 -2
  17. package/dist/tool-handlers/generate-audio.js +2 -4
  18. package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
  19. package/dist/tool-handlers/generate-image-dedicated.js +176 -0
  20. package/dist/tool-handlers/generate-image-input.d.ts +3 -0
  21. package/dist/tool-handlers/generate-image-input.js +38 -0
  22. package/dist/tool-handlers/generate-image.d.ts +13 -51
  23. package/dist/tool-handlers/generate-image.js +32 -119
  24. package/dist/tool-handlers/generate-video.d.ts +2 -2
  25. package/dist/tool-handlers/generate-video.js +78 -30
  26. package/dist/tool-handlers/image-utils.d.ts +1 -0
  27. package/dist/tool-handlers/image-utils.js +26 -16
  28. package/dist/tool-handlers/openrouter-errors.js +6 -2
  29. package/dist/tool-handlers/path-safety.js +32 -5
  30. package/dist/tool-handlers/provider-routing.js +7 -2
  31. package/dist/tool-handlers/rerank.js +2 -5
  32. package/dist/tool-handlers/search-models.d.ts +2 -2
  33. package/dist/tool-handlers/search-models.js +2 -6
  34. package/dist/tool-handlers/speech-to-text.d.ts +20 -0
  35. package/dist/tool-handlers/speech-to-text.js +140 -0
  36. package/dist/tool-handlers/structured-output.d.ts +8 -0
  37. package/dist/tool-handlers/structured-output.js +11 -0
  38. package/dist/tool-handlers/text-to-speech.d.ts +29 -0
  39. package/dist/tool-handlers/text-to-speech.js +105 -0
  40. package/dist/tool-handlers/video-utils.js +6 -9
  41. package/dist/tool-handlers.js +253 -125
  42. package/dist/version.d.ts +1 -1
  43. package/dist/version.js +1 -1
  44. package/package.json +27 -15
@@ -0,0 +1,51 @@
1
+ import OpenAI from 'openai';
2
+ import type { ChatCompletionMessageParam } from 'openai/resources/chat/completions.js';
3
+ import { type ProviderRoutingOptions } from './provider-routing.js';
4
+ import { type CacheOptions } from './cache.js';
5
+ export interface StartChatCompletionRequest extends CacheOptions {
6
+ messages: ChatCompletionMessageParam[];
7
+ model?: string;
8
+ temperature?: number;
9
+ max_tokens?: number;
10
+ provider?: ProviderRoutingOptions;
11
+ include_reasoning?: boolean;
12
+ online?: boolean;
13
+ web_max_results?: number;
14
+ }
15
+ export interface GetChatCompletionStatusRequest {
16
+ job_id: string;
17
+ }
18
+ export type AsyncJobStatus = 'queued' | 'running' | 'completed' | 'failed';
19
+ export declare function handleStartChatCompletion(request: {
20
+ params: {
21
+ arguments: StartChatCompletionRequest;
22
+ };
23
+ }, openai: OpenAI, defaultModel?: string): Promise<import("../errors.js").ToolErrorResult | {
24
+ content: {
25
+ type: "text";
26
+ text: string;
27
+ }[];
28
+ _meta: {
29
+ server_version: string;
30
+ job_id: string;
31
+ status: "running";
32
+ model: string;
33
+ };
34
+ }>;
35
+ export declare function handleGetChatCompletionStatus(request: {
36
+ params: {
37
+ arguments: GetChatCompletionStatusRequest;
38
+ };
39
+ }): Promise<import("../errors.js").ToolErrorResult | {
40
+ content: {
41
+ type: "text";
42
+ text: string;
43
+ }[];
44
+ _meta: {
45
+ server_version: string;
46
+ job_id: string;
47
+ status: "queued" | "completed" | "running";
48
+ model: string;
49
+ created_at: string;
50
+ };
51
+ }>;
@@ -0,0 +1,216 @@
1
+ /**
2
+ * Async chat completions — resumable workflow for long-running requests.
3
+ *
4
+ * Problem: Remote MCP bridges (Cowork, etc.) kill tool calls after ~60s.
5
+ * Reasoning models can take much longer. Unlike video, `chat_completion`
6
+ * currently has no background job mechanism.
7
+ *
8
+ * Solution: Two tools that mirror the video pattern:
9
+ * - `start_chat_completion` — fires off the request in the background,
10
+ * returns a `job_id` immediately.
11
+ * - `get_chat_completion_status` — returns queued/running/completed/failed,
12
+ * with the final response on completion.
13
+ *
14
+ * Job state is held in memory (survives within a single MCP session).
15
+ * Optionally persisted to OPENROUTER_OUTPUT_DIR/openrouter-jobs/ for
16
+ * crash recovery.
17
+ */
18
+ import { promises as fs } from 'fs';
19
+ import path from 'node:path';
20
+ import { ErrorCode, toolError } from '../errors.js';
21
+ import { SERVER_VERSION } from '../version.js';
22
+ import { logger } from '../logger.js';
23
+ import { extractCompletionText, buildCompletionMeta, } from './completion-utils.js';
24
+ import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
25
+ import { buildCacheHeaders } from './cache.js';
26
+ // ─── Job Store ───────────────────────────────────────────────────────────────
27
+ const jobs = new Map();
28
+ let jobCounter = 0;
29
+ function generateJobId() {
30
+ jobCounter += 1;
31
+ const ts = new Date().toISOString().replace(/[-:T]/g, '').slice(0, 14);
32
+ return `chat_${ts}_${String(jobCounter).padStart(3, '0')}`;
33
+ }
34
+ function getJobsDir() {
35
+ const outputDir = process.env.OPENROUTER_OUTPUT_DIR;
36
+ if (!outputDir)
37
+ return null;
38
+ return path.join(outputDir, 'openrouter-jobs');
39
+ }
40
+ async function persistJob(job) {
41
+ const dir = getJobsDir();
42
+ if (!dir)
43
+ return;
44
+ try {
45
+ const jobDir = path.join(dir, job.id);
46
+ await fs.mkdir(jobDir, { recursive: true });
47
+ await fs.writeFile(path.join(jobDir, 'status.json'), JSON.stringify(job, null, 2));
48
+ if (job.status === 'completed' && job.result?.text) {
49
+ await fs.writeFile(path.join(jobDir, 'response.md'), job.result.text);
50
+ }
51
+ }
52
+ catch (err) {
53
+ logger.warn('async_chat.persist_error', {
54
+ job_id: job.id,
55
+ err: err instanceof Error ? err.message : String(err),
56
+ });
57
+ }
58
+ }
59
+ // ─── Handlers ────────────────────────────────────────────────────────────────
60
+ const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
61
+ function readIncludeReasoningDefault() {
62
+ const raw = (process.env.OPENROUTER_INCLUDE_REASONING ?? '').trim().toLowerCase();
63
+ return raw === '1' || raw === 'true' || raw === 'yes';
64
+ }
65
+ export async function handleStartChatCompletion(request, openai, defaultModel) {
66
+ const args = request.params.arguments ?? { messages: [] };
67
+ const { messages, model, temperature, max_tokens, provider, include_reasoning, online, web_max_results, cache, cache_ttl, cache_clear, } = args;
68
+ if (!messages?.length) {
69
+ return toolError(ErrorCode.INVALID_INPUT, 'Messages array cannot be empty.');
70
+ }
71
+ const effectiveModel = model || defaultModel || DEFAULT_MODEL;
72
+ const jobId = generateJobId();
73
+ // Create the job immediately
74
+ const job = {
75
+ id: jobId,
76
+ status: 'running',
77
+ createdAt: new Date().toISOString(),
78
+ model: effectiveModel,
79
+ };
80
+ jobs.set(jobId, job);
81
+ logger.audit('async_chat.start', {
82
+ job_id: jobId,
83
+ model: effectiveModel,
84
+ message_count: messages.length,
85
+ });
86
+ // Fire and forget — the completion runs in the background
87
+ runCompletionInBackground(job, openai, {
88
+ messages,
89
+ model: effectiveModel,
90
+ temperature,
91
+ max_tokens,
92
+ provider,
93
+ include_reasoning,
94
+ online,
95
+ web_max_results,
96
+ cache,
97
+ cache_ttl,
98
+ cache_clear,
99
+ });
100
+ // Return immediately with the job ID
101
+ return {
102
+ content: [
103
+ {
104
+ type: 'text',
105
+ text: `Chat completion job started. Use get_chat_completion_status with job_id="${jobId}" to check results.`,
106
+ },
107
+ ],
108
+ _meta: {
109
+ server_version: SERVER_VERSION,
110
+ job_id: jobId,
111
+ status: 'running',
112
+ model: effectiveModel,
113
+ },
114
+ };
115
+ }
116
+ async function runCompletionInBackground(job, openai, opts) {
117
+ const providerOptions = mergeProviderOptions(readProviderDefaults(), opts.provider);
118
+ const providerBody = buildProviderBody(providerOptions);
119
+ const effectiveMaxTokens = resolveMaxTokens(opts.max_tokens);
120
+ const wantsReasoning = opts.include_reasoning ?? readIncludeReasoningDefault();
121
+ const body = {
122
+ model: opts.model,
123
+ messages: opts.messages,
124
+ temperature: opts.temperature ?? 1,
125
+ };
126
+ if (typeof effectiveMaxTokens === 'number')
127
+ body.max_tokens = effectiveMaxTokens;
128
+ if (providerBody)
129
+ body.provider = providerBody;
130
+ if (wantsReasoning)
131
+ body.include_reasoning = true;
132
+ if (opts.online) {
133
+ const plugin = { id: 'web' };
134
+ if (typeof opts.web_max_results === 'number' && opts.web_max_results > 0) {
135
+ plugin.max_results = opts.web_max_results;
136
+ }
137
+ body.plugins = [plugin];
138
+ }
139
+ const headers = buildCacheHeaders({
140
+ cache: opts.cache,
141
+ cache_ttl: opts.cache_ttl,
142
+ cache_clear: opts.cache_clear,
143
+ });
144
+ const requestOpts = Object.keys(headers).length > 0 ? { headers } : undefined;
145
+ try {
146
+ const completion = (await openai.chat.completions.create(body, requestOpts));
147
+ const extracted = extractCompletionText(completion);
148
+ if (!extracted.text) {
149
+ job.status = 'failed';
150
+ job.error = 'Model returned no textual content.';
151
+ }
152
+ else {
153
+ job.status = 'completed';
154
+ job.result = {
155
+ text: extracted.text,
156
+ meta: buildCompletionMeta(extracted, {
157
+ includeReasoning: wantsReasoning,
158
+ extra: { server_version: SERVER_VERSION },
159
+ }),
160
+ };
161
+ }
162
+ }
163
+ catch (err) {
164
+ job.status = 'failed';
165
+ job.error = err instanceof Error ? err.message : String(err);
166
+ logger.warn('async_chat.failed', { job_id: job.id, error: job.error });
167
+ }
168
+ await persistJob(job);
169
+ }
170
+ export async function handleGetChatCompletionStatus(request) {
171
+ const args = request.params.arguments ?? {};
172
+ const jobId = args.job_id?.trim();
173
+ if (!jobId) {
174
+ return toolError(ErrorCode.INVALID_INPUT, 'job_id is required.');
175
+ }
176
+ const job = jobs.get(jobId);
177
+ if (!job) {
178
+ return toolError(ErrorCode.INVALID_INPUT, `No job found with id "${jobId}". Jobs are stored in memory for the current session only.`);
179
+ }
180
+ if (job.status === 'completed' && job.result) {
181
+ return {
182
+ content: [{ type: 'text', text: job.result.text }],
183
+ _meta: {
184
+ server_version: SERVER_VERSION,
185
+ job_id: jobId,
186
+ status: 'completed',
187
+ model: job.model,
188
+ created_at: job.createdAt,
189
+ ...job.result.meta,
190
+ },
191
+ };
192
+ }
193
+ if (job.status === 'failed') {
194
+ return toolError(ErrorCode.JOB_FAILED, job.error || 'Job failed.', {
195
+ job_id: jobId,
196
+ model: job.model,
197
+ created_at: job.createdAt,
198
+ });
199
+ }
200
+ // Still running
201
+ return {
202
+ content: [
203
+ {
204
+ type: 'text',
205
+ text: `Job ${jobId} is still ${job.status}. Try again in a few seconds.`,
206
+ },
207
+ ],
208
+ _meta: {
209
+ server_version: SERVER_VERSION,
210
+ job_id: jobId,
211
+ status: job.status,
212
+ model: job.model,
213
+ created_at: job.createdAt,
214
+ },
215
+ };
216
+ }
@@ -5,6 +5,7 @@
5
5
  import path from 'path';
6
6
  import { promises as fs } from 'fs';
7
7
  import { readEnvInt, fetchHttpResource, parseBase64DataUrl } from './fetch-utils.js';
8
+ import { resolveSafeInputPath } from './path-safety.js';
8
9
  // Re-export for tests
9
10
  export { isBlockedIPv4, assertUrlSafeForFetch } from './fetch-utils.js';
10
11
  const DEFAULT_FETCH_TIMEOUT_MS = 30_000;
@@ -117,10 +118,11 @@ export async function prepareAudioData(source) {
117
118
  return { data: buffer.toString('base64'), format };
118
119
  }
119
120
  // --- local file ---
120
- const format = getAudioFormat(source);
121
+ const safe = await resolveSafeInputPath(source);
122
+ const format = getAudioFormat(safe);
121
123
  if (!format) {
122
124
  throw new Error(`Unsupported audio format for file: ${source}. Supported: ${SUPPORTED_AUDIO_FORMATS.join(', ')}`);
123
125
  }
124
- const buffer = await fs.readFile(source);
126
+ const buffer = await fs.readFile(safe);
125
127
  return { data: buffer.toString('base64'), format };
126
128
  }
@@ -3,7 +3,7 @@ import { SERVER_VERSION } from '../version.js';
3
3
  import { classifyUpstreamError } from './openrouter-errors.js';
4
4
  import { extractCompletionText, detectReasoningCutoff, buildCompletionMeta, } from './completion-utils.js';
5
5
  import { readProviderDefaults, mergeProviderOptions, buildProviderBody, resolveMaxTokens, } from './provider-routing.js';
6
- import { buildCacheHeaders, extractCacheMeta, } from './cache.js';
6
+ import { buildCacheHeaders, extractCacheMeta } from './cache.js';
7
7
  import { awaitCompletionWithHeaders } from './openai-withresponse.js';
8
8
  const DEFAULT_MODEL = 'nvidia/nemotron-nano-12b-v2-vl:free';
9
9
  function readIncludeReasoningDefault() {
@@ -151,11 +151,25 @@ export function isBlockedIPv6(ip) {
151
151
  return false;
152
152
  const [g0, g1, g2, g3, g4, g5, g6, g7] = groups;
153
153
  // :: (unspecified)
154
- if (g0 === 0 && g1 === 0 && g2 === 0 && g3 === 0 && g4 === 0 && g5 === 0 && g6 === 0 && g7 === 0) {
154
+ if (g0 === 0 &&
155
+ g1 === 0 &&
156
+ g2 === 0 &&
157
+ g3 === 0 &&
158
+ g4 === 0 &&
159
+ g5 === 0 &&
160
+ g6 === 0 &&
161
+ g7 === 0) {
155
162
  return true;
156
163
  }
157
164
  // ::1 (loopback)
158
- if (g0 === 0 && g1 === 0 && g2 === 0 && g3 === 0 && g4 === 0 && g5 === 0 && g6 === 0 && g7 === 1) {
165
+ if (g0 === 0 &&
166
+ g1 === 0 &&
167
+ g2 === 0 &&
168
+ g3 === 0 &&
169
+ g4 === 0 &&
170
+ g5 === 0 &&
171
+ g6 === 0 &&
172
+ g7 === 1) {
159
173
  return true;
160
174
  }
161
175
  // ::ffff:0:0/96 — IPv4-mapped. Re-check the embedded IPv4.
@@ -95,9 +95,7 @@ export async function handleGenerateAudio(request, openai) {
95
95
  logger.audit('generate_audio.start', {
96
96
  model: model || DEFAULT_MODEL,
97
97
  voice: voice?.trim() || DEFAULT_VOICE,
98
- format: VALID_FORMATS.includes(format ?? '')
99
- ? format
100
- : DEFAULT_FORMAT,
98
+ format: VALID_FORMATS.includes(format ?? '') ? format : DEFAULT_FORMAT,
101
99
  prompt_preview: prompt.slice(0, 80),
102
100
  save_path: save_path ? 'provided' : 'none',
103
101
  });
@@ -156,7 +154,7 @@ export async function handleGenerateAudio(request, openai) {
156
154
  const detected = detectAudioFormat(audioBuffer);
157
155
  // Always wrap raw PCM in WAV so it's playable
158
156
  if (detected.ext === 'pcm') {
159
- audioBuffer = wrapPcmInWav(audioBuffer);
157
+ audioBuffer = Buffer.from(wrapPcmInWav(audioBuffer));
160
158
  detected.ext = 'wav';
161
159
  detected.mimeType = 'audio/wav';
162
160
  }
@@ -0,0 +1,32 @@
1
+ import type { OpenRouterAPIClient } from '../openrouter-api.js';
2
+ import { type CacheOptions } from './cache.js';
3
+ export interface GenerateImageDedicatedRequest extends CacheOptions {
4
+ prompt: string;
5
+ model?: string;
6
+ resolution?: string;
7
+ aspect_ratio?: string;
8
+ quality?: string;
9
+ output_format?: string;
10
+ n?: number;
11
+ input_references?: string[];
12
+ save_path?: string;
13
+ provider?: Record<string, unknown>;
14
+ }
15
+ export declare function handleGenerateImageDedicated(request: {
16
+ params: {
17
+ arguments: GenerateImageDedicatedRequest;
18
+ };
19
+ }, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
20
+ content: ({
21
+ type: "text";
22
+ text: string;
23
+ mimeType?: undefined;
24
+ data?: undefined;
25
+ } | {
26
+ type: "image";
27
+ mimeType: string;
28
+ data: string;
29
+ text?: undefined;
30
+ })[];
31
+ _meta: Record<string, unknown>;
32
+ }>;
@@ -0,0 +1,176 @@
1
+ /**
2
+ * generate_image_dedicated — uses OpenRouter's dedicated POST /api/v1/images
3
+ * endpoint (launched June 2026) for image generation. Supports normalized
4
+ * resolution tiers, aspect ratios, quality levels, output formats, and
5
+ * input_references for image-to-image workflows.
6
+ *
7
+ * This is distinct from the original `generate_image` tool which uses chat
8
+ * completions with `modalities: ['image', 'text']`. New image models are
9
+ * added exclusively to this dedicated endpoint.
10
+ */
11
+ import { promises as fs } from 'fs';
12
+ import path from 'node:path';
13
+ import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
14
+ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
15
+ import { SERVER_VERSION } from '../version.js';
16
+ import { logger } from '../logger.js';
17
+ import { classifyUpstreamError } from './openrouter-errors.js';
18
+ import { buildCacheHeaders } from './cache.js';
19
+ const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
20
+ const VALID_RESOLUTIONS = new Set(['512', '0.5K', '1K', '2K', '4K']);
21
+ const VALID_QUALITIES = new Set(['auto', 'low', 'medium', 'high']);
22
+ const VALID_OUTPUT_FORMATS = new Set(['png', 'jpeg', 'webp', 'svg']);
23
+ /**
24
+ * Resolve an input image reference (local path, URL, or data URL) into the
25
+ * OpenRouter `input_references` shape: `{ type: "image_url", image_url: { url } }`.
26
+ */
27
+ async function resolveReference(source) {
28
+ const trimmed = source.trim();
29
+ if (!trimmed)
30
+ throw new Error('Empty input_references entry');
31
+ // Data URLs and HTTP URLs pass through directly
32
+ if (trimmed.startsWith('data:') || /^https?:\/\//i.test(trimmed)) {
33
+ return { type: 'image_url', image_url: { url: trimmed } };
34
+ }
35
+ // Local file: sandbox, read, and convert to data URL
36
+ const abs = await resolveSafeInputPath(trimmed);
37
+ const buf = await fs.readFile(abs);
38
+ const ext = path.extname(abs).toLowerCase();
39
+ const mime = ext === '.png' ? 'image/png' :
40
+ ext === '.webp' ? 'image/webp' :
41
+ ext === '.gif' ? 'image/gif' :
42
+ ext === '.svg' ? 'image/svg+xml' :
43
+ 'image/jpeg';
44
+ const dataUrl = `data:${mime};base64,${buf.toString('base64')}`;
45
+ return { type: 'image_url', image_url: { url: dataUrl } };
46
+ }
47
+ export async function handleGenerateImageDedicated(request, apiClient) {
48
+ const args = request.params.arguments ?? {};
49
+ const { prompt, model, resolution, aspect_ratio, quality, output_format, n, input_references, save_path, provider, cache, cache_ttl, cache_clear, } = args;
50
+ if (!prompt?.trim()) {
51
+ return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
52
+ }
53
+ logger.audit('generate_image_dedicated.start', {
54
+ model: model || DEFAULT_MODEL,
55
+ prompt_preview: prompt.slice(0, 80),
56
+ resolution,
57
+ aspect_ratio,
58
+ quality,
59
+ output_format,
60
+ input_references_count: input_references?.length ?? 0,
61
+ save_path: save_path ? 'provided' : 'none',
62
+ });
63
+ // Validate enums
64
+ if (resolution && !VALID_RESOLUTIONS.has(resolution)) {
65
+ return toolError(ErrorCode.INVALID_INPUT, `resolution '${resolution}' is not supported. Valid: ${[...VALID_RESOLUTIONS].join(', ')}.`);
66
+ }
67
+ if (quality && !VALID_QUALITIES.has(quality)) {
68
+ return toolError(ErrorCode.INVALID_INPUT, `quality '${quality}' is not supported. Valid: ${[...VALID_QUALITIES].join(', ')}.`);
69
+ }
70
+ if (output_format && !VALID_OUTPUT_FORMATS.has(output_format)) {
71
+ return toolError(ErrorCode.INVALID_INPUT, `output_format '${output_format}' is not supported. Valid: ${[...VALID_OUTPUT_FORMATS].join(', ')}.`);
72
+ }
73
+ // Resolve save path early
74
+ let safeSavePath = null;
75
+ if (save_path) {
76
+ try {
77
+ safeSavePath = await resolveSafeOutputPath(save_path);
78
+ }
79
+ catch (err) {
80
+ if (err instanceof UnsafeOutputPathError)
81
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
82
+ return toolErrorFrom(ErrorCode.INTERNAL, err);
83
+ }
84
+ }
85
+ // Build request body
86
+ const body = {
87
+ model: model || DEFAULT_MODEL,
88
+ prompt,
89
+ };
90
+ if (resolution)
91
+ body.resolution = resolution;
92
+ if (aspect_ratio)
93
+ body.aspect_ratio = aspect_ratio;
94
+ if (quality)
95
+ body.quality = quality;
96
+ if (output_format)
97
+ body.output_format = output_format;
98
+ if (typeof n === 'number' && n > 0)
99
+ body.n = n;
100
+ if (provider && typeof provider === 'object')
101
+ body.provider = provider;
102
+ // Resolve input references
103
+ if (input_references?.length) {
104
+ try {
105
+ const refs = await Promise.all(input_references.map(resolveReference));
106
+ body.input_references = refs;
107
+ }
108
+ catch (err) {
109
+ if (err instanceof UnsafeOutputPathError)
110
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
111
+ return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_references');
112
+ }
113
+ }
114
+ // Build cache headers
115
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
116
+ let response;
117
+ try {
118
+ response = await apiClient.generateImage(body, headers);
119
+ }
120
+ catch (err) {
121
+ return classifyUpstreamError(err, 'generate_image_dedicated');
122
+ }
123
+ const images = response.data ?? [];
124
+ if (!images.length || (!images[0]?.b64_json && !images[0]?.url)) {
125
+ return toolError(ErrorCode.UPSTREAM_REFUSED, 'Model returned no image data.', {
126
+ response_keys: Object.keys(response),
127
+ });
128
+ }
129
+ const firstImage = images[0];
130
+ const imageData = firstImage.b64_json;
131
+ const mimeType = output_format === 'png' ? 'image/png' :
132
+ output_format === 'webp' ? 'image/webp' :
133
+ output_format === 'svg' ? 'image/svg+xml' :
134
+ output_format === 'jpeg' ? 'image/jpeg' :
135
+ 'image/png'; // default
136
+ const baseMeta = {
137
+ server_version: SERVER_VERSION,
138
+ model: model || DEFAULT_MODEL,
139
+ images_count: images.length,
140
+ };
141
+ if (response.usage)
142
+ baseMeta.usage = response.usage;
143
+ if (firstImage.revised_prompt)
144
+ baseMeta.revised_prompt = firstImage.revised_prompt;
145
+ // Save to file if requested
146
+ if (safeSavePath && imageData) {
147
+ try {
148
+ await fs.writeFile(safeSavePath, imageData, { encoding: 'base64' });
149
+ }
150
+ catch (err) {
151
+ return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
152
+ }
153
+ baseMeta.save_path = safeSavePath;
154
+ return {
155
+ content: [
156
+ { type: 'text', text: `Image saved to: ${safeSavePath}` },
157
+ ...(imageData ? [{ type: 'image', mimeType, data: imageData }] : []),
158
+ ],
159
+ _meta: baseMeta,
160
+ };
161
+ }
162
+ // Return inline
163
+ if (imageData) {
164
+ return {
165
+ content: [{ type: 'image', mimeType, data: imageData }],
166
+ _meta: baseMeta,
167
+ };
168
+ }
169
+ // URL-only response (some models return URLs instead of base64)
170
+ return {
171
+ content: [
172
+ { type: 'text', text: `Image generated. URL: ${firstImage.url}` },
173
+ ],
174
+ _meta: { ...baseMeta, image_url: firstImage.url },
175
+ };
176
+ }
@@ -0,0 +1,3 @@
1
+ import OpenAI from 'openai';
2
+ export declare function resolveInputImage(ref: string): Promise<string>;
3
+ export declare function buildUserContent(prompt: string, inputImages?: string[]): Promise<string | OpenAI.Chat.Completions.ChatCompletionContentPart[]>;
@@ -0,0 +1,38 @@
1
+ import { promises as fs } from 'fs';
2
+ import path from 'node:path';
3
+ import { resolveSafeInputPath } from './path-safety.js';
4
+ import { mimeFromExtension } from './image-utils.js';
5
+ export async function resolveInputImage(ref) {
6
+ const trimmed = ref.trim();
7
+ if (!trimmed)
8
+ throw new Error('empty input_images entry');
9
+ if (trimmed.startsWith('data:'))
10
+ return trimmed;
11
+ if (/^https?:\/\//i.test(trimmed))
12
+ return trimmed;
13
+ const abs = await resolveSafeInputPath(trimmed);
14
+ const buf = await fs.readFile(abs);
15
+ const mime = mimeFromExtension(path.extname(abs)) || 'image/png';
16
+ return `data:${mime};base64,${buf.toString('base64')}`;
17
+ }
18
+ export async function buildUserContent(prompt, inputImages) {
19
+ if (!inputImages?.length) {
20
+ return `Generate an image: ${prompt}`;
21
+ }
22
+ const parts = [
23
+ {
24
+ type: 'text',
25
+ text: `Generate an image based on this prompt, using the following reference image(s) ` +
26
+ `for visual consistency. Match the appearance, identity, and style of the references ` +
27
+ `closely; do not alter them.\n\nPrompt: ${prompt}`,
28
+ },
29
+ ];
30
+ const urls = await Promise.all(inputImages.map((ref) => resolveInputImage(ref)));
31
+ for (const url of urls) {
32
+ parts.push({
33
+ type: 'image_url',
34
+ image_url: { url, detail: 'high' },
35
+ });
36
+ }
37
+ return parts;
38
+ }