@stabgan/openrouter-mcp-multimodal 4.5.3 → 4.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,176 @@
1
+ /**
2
+ * generate_image_dedicated — uses OpenRouter's dedicated POST /api/v1/images
3
+ * endpoint (launched June 2026) for image generation. Supports normalized
4
+ * resolution tiers, aspect ratios, quality levels, output formats, and
5
+ * input_references for image-to-image workflows.
6
+ *
7
+ * This is distinct from the original `generate_image` tool which uses chat
8
+ * completions with `modalities: ['image', 'text']`. New image models are
9
+ * added exclusively to this dedicated endpoint.
10
+ */
11
+ import { promises as fs } from 'fs';
12
+ import path from 'node:path';
13
+ import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
14
+ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
15
+ import { SERVER_VERSION } from '../version.js';
16
+ import { logger } from '../logger.js';
17
+ import { classifyUpstreamError } from './openrouter-errors.js';
18
+ import { buildCacheHeaders } from './cache.js';
19
+ const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
20
+ const VALID_RESOLUTIONS = new Set(['512', '0.5K', '1K', '2K', '4K']);
21
+ const VALID_QUALITIES = new Set(['auto', 'low', 'medium', 'high']);
22
+ const VALID_OUTPUT_FORMATS = new Set(['png', 'jpeg', 'webp', 'svg']);
23
+ /**
24
+ * Resolve an input image reference (local path, URL, or data URL) into the
25
+ * OpenRouter `input_references` shape: `{ type: "image_url", image_url: { url } }`.
26
+ */
27
+ async function resolveReference(source) {
28
+ const trimmed = source.trim();
29
+ if (!trimmed)
30
+ throw new Error('Empty input_references entry');
31
+ // Data URLs and HTTP URLs pass through directly
32
+ if (trimmed.startsWith('data:') || /^https?:\/\//i.test(trimmed)) {
33
+ return { type: 'image_url', image_url: { url: trimmed } };
34
+ }
35
+ // Local file: sandbox, read, and convert to data URL
36
+ const abs = await resolveSafeInputPath(trimmed);
37
+ const buf = await fs.readFile(abs);
38
+ const ext = path.extname(abs).toLowerCase();
39
+ const mime = ext === '.png' ? 'image/png' :
40
+ ext === '.webp' ? 'image/webp' :
41
+ ext === '.gif' ? 'image/gif' :
42
+ ext === '.svg' ? 'image/svg+xml' :
43
+ 'image/jpeg';
44
+ const dataUrl = `data:${mime};base64,${buf.toString('base64')}`;
45
+ return { type: 'image_url', image_url: { url: dataUrl } };
46
+ }
47
+ export async function handleGenerateImageDedicated(request, apiClient) {
48
+ const args = request.params.arguments ?? {};
49
+ const { prompt, model, resolution, aspect_ratio, quality, output_format, n, input_references, save_path, provider, cache, cache_ttl, cache_clear, } = args;
50
+ if (!prompt?.trim()) {
51
+ return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
52
+ }
53
+ logger.audit('generate_image_dedicated.start', {
54
+ model: model || DEFAULT_MODEL,
55
+ prompt_preview: prompt.slice(0, 80),
56
+ resolution,
57
+ aspect_ratio,
58
+ quality,
59
+ output_format,
60
+ input_references_count: input_references?.length ?? 0,
61
+ save_path: save_path ? 'provided' : 'none',
62
+ });
63
+ // Validate enums
64
+ if (resolution && !VALID_RESOLUTIONS.has(resolution)) {
65
+ return toolError(ErrorCode.INVALID_INPUT, `resolution '${resolution}' is not supported. Valid: ${[...VALID_RESOLUTIONS].join(', ')}.`);
66
+ }
67
+ if (quality && !VALID_QUALITIES.has(quality)) {
68
+ return toolError(ErrorCode.INVALID_INPUT, `quality '${quality}' is not supported. Valid: ${[...VALID_QUALITIES].join(', ')}.`);
69
+ }
70
+ if (output_format && !VALID_OUTPUT_FORMATS.has(output_format)) {
71
+ return toolError(ErrorCode.INVALID_INPUT, `output_format '${output_format}' is not supported. Valid: ${[...VALID_OUTPUT_FORMATS].join(', ')}.`);
72
+ }
73
+ // Resolve save path early
74
+ let safeSavePath = null;
75
+ if (save_path) {
76
+ try {
77
+ safeSavePath = await resolveSafeOutputPath(save_path);
78
+ }
79
+ catch (err) {
80
+ if (err instanceof UnsafeOutputPathError)
81
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
82
+ return toolErrorFrom(ErrorCode.INTERNAL, err);
83
+ }
84
+ }
85
+ // Build request body
86
+ const body = {
87
+ model: model || DEFAULT_MODEL,
88
+ prompt,
89
+ };
90
+ if (resolution)
91
+ body.resolution = resolution;
92
+ if (aspect_ratio)
93
+ body.aspect_ratio = aspect_ratio;
94
+ if (quality)
95
+ body.quality = quality;
96
+ if (output_format)
97
+ body.output_format = output_format;
98
+ if (typeof n === 'number' && n > 0)
99
+ body.n = n;
100
+ if (provider && typeof provider === 'object')
101
+ body.provider = provider;
102
+ // Resolve input references
103
+ if (input_references?.length) {
104
+ try {
105
+ const refs = await Promise.all(input_references.map(resolveReference));
106
+ body.input_references = refs;
107
+ }
108
+ catch (err) {
109
+ if (err instanceof UnsafeOutputPathError)
110
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
111
+ return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_references');
112
+ }
113
+ }
114
+ // Build cache headers
115
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
116
+ let response;
117
+ try {
118
+ response = await apiClient.generateImage(body, headers);
119
+ }
120
+ catch (err) {
121
+ return classifyUpstreamError(err, 'generate_image_dedicated');
122
+ }
123
+ const images = response.data ?? [];
124
+ if (!images.length || (!images[0]?.b64_json && !images[0]?.url)) {
125
+ return toolError(ErrorCode.UPSTREAM_REFUSED, 'Model returned no image data.', {
126
+ response_keys: Object.keys(response),
127
+ });
128
+ }
129
+ const firstImage = images[0];
130
+ const imageData = firstImage.b64_json;
131
+ const mimeType = output_format === 'png' ? 'image/png' :
132
+ output_format === 'webp' ? 'image/webp' :
133
+ output_format === 'svg' ? 'image/svg+xml' :
134
+ output_format === 'jpeg' ? 'image/jpeg' :
135
+ 'image/png'; // default
136
+ const baseMeta = {
137
+ server_version: SERVER_VERSION,
138
+ model: model || DEFAULT_MODEL,
139
+ images_count: images.length,
140
+ };
141
+ if (response.usage)
142
+ baseMeta.usage = response.usage;
143
+ if (firstImage.revised_prompt)
144
+ baseMeta.revised_prompt = firstImage.revised_prompt;
145
+ // Save to file if requested
146
+ if (safeSavePath && imageData) {
147
+ try {
148
+ await fs.writeFile(safeSavePath, imageData, { encoding: 'base64' });
149
+ }
150
+ catch (err) {
151
+ return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
152
+ }
153
+ baseMeta.save_path = safeSavePath;
154
+ return {
155
+ content: [
156
+ { type: 'text', text: `Image saved to: ${safeSavePath}` },
157
+ ...(imageData ? [{ type: 'image', mimeType, data: imageData }] : []),
158
+ ],
159
+ _meta: baseMeta,
160
+ };
161
+ }
162
+ // Return inline
163
+ if (imageData) {
164
+ return {
165
+ content: [{ type: 'image', mimeType, data: imageData }],
166
+ _meta: baseMeta,
167
+ };
168
+ }
169
+ // URL-only response (some models return URLs instead of base64)
170
+ return {
171
+ content: [
172
+ { type: 'text', text: `Image generated. URL: ${firstImage.url}` },
173
+ ],
174
+ _meta: { ...baseMeta, image_url: firstImage.url },
175
+ };
176
+ }
@@ -34,7 +34,7 @@ export declare function handleGenerateVideo(request: {
34
34
  };
35
35
  }, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
36
36
  content: {
37
- type: "text";
37
+ type: string;
38
38
  text: string;
39
39
  }[];
40
40
  isError: false;
@@ -97,7 +97,7 @@ export declare function handleGenerateVideoFromImage(request: {
97
97
  };
98
98
  }, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
99
99
  content: {
100
- type: "text";
100
+ type: string;
101
101
  text: string;
102
102
  }[];
103
103
  isError: false;
@@ -11,6 +11,34 @@ const DEFAULT_POLL_INTERVAL_MS = 15_000;
11
11
  const DEFAULT_MAX_WAIT_MS = 10 * 60_000;
12
12
  const MIN_POLL_INTERVAL_MS = 50; // just to avoid a 0ms busy-loop if a caller omits
13
13
  const INLINE_RETURN_CEILING_BYTES = 10 * 1024 * 1024;
14
+ /** Models deprecated by OpenAI — removal date: 2026-09-24. */
15
+ const SORA_DEPRECATED_MODELS = new Set([
16
+ 'openai/sora-2',
17
+ 'openai/sora-2-pro',
18
+ 'openai/sora-2-2025-10-06',
19
+ 'openai/sora-2-2025-12-08',
20
+ 'openai/sora-2-pro-2025-10-06',
21
+ ]);
22
+ const SORA_ALTERNATIVES = [
23
+ 'google/veo-3.1 (recommended — fast, audio support)',
24
+ 'google/veo-3.1-fast (budget-friendly)',
25
+ 'bytedance/seedance-2.0 (high quality)',
26
+ 'bytedance/seedance-2.0-fast (fast turnaround)',
27
+ 'alibaba/wan-2.7 (good for artistic styles)',
28
+ ];
29
+ /**
30
+ * Check if the model is a deprecated Sora model and return a warning string,
31
+ * or null if no deprecation applies.
32
+ */
33
+ function checkSoraDeprecation(model) {
34
+ const normalized = model.toLowerCase().trim();
35
+ if (!SORA_DEPRECATED_MODELS.has(normalized) && !normalized.startsWith('openai/sora')) {
36
+ return null;
37
+ }
38
+ return (`⚠️ DEPRECATION WARNING: ${model} is deprecated by OpenAI and will be removed from the API on September 24, 2026. ` +
39
+ `Your request will still be attempted, but may fail. Recommended alternatives:\n` +
40
+ SORA_ALTERNATIVES.map((a) => ` • ${a}`).join('\n'));
41
+ }
14
42
  function getMaxInlineBytes() {
15
43
  return readEnvInt('OPENROUTER_VIDEO_INLINE_MAX_BYTES', INLINE_RETURN_CEILING_BYTES, 4096);
16
44
  }
@@ -94,8 +122,9 @@ async function attachFrameImages(args, body) {
94
122
  ? {
95
123
  kind: 'frame',
96
124
  entry: {
125
+ type: 'image_url',
126
+ image_url: { url: `data:${img.mime};base64,${img.data}` },
97
127
  frame_type: 'first_frame',
98
- image: { url: `data:${img.mime};base64,${img.data}` },
99
128
  },
100
129
  }
101
130
  : null));
@@ -105,8 +134,9 @@ async function attachFrameImages(args, body) {
105
134
  ? {
106
135
  kind: 'frame',
107
136
  entry: {
137
+ type: 'image_url',
138
+ image_url: { url: `data:${img.mime};base64,${img.data}` },
108
139
  frame_type: 'last_frame',
109
- image: { url: `data:${img.mime};base64,${img.data}` },
110
140
  },
111
141
  }
112
142
  : null));
@@ -121,7 +151,10 @@ async function attachFrameImages(args, body) {
121
151
  const refResults = await Promise.all(args.reference_images.map((src) => prepareImageInput(src)));
122
152
  const refs = refResults
123
153
  .filter((img) => img !== null)
124
- .map((img) => ({ image: { url: `data:${img.mime};base64,${img.data}` } }));
154
+ .map((img) => ({
155
+ type: 'image_url',
156
+ image_url: { url: `data:${img.mime};base64,${img.data}` },
157
+ }));
125
158
  if (refs.length)
126
159
  body.input_references = refs;
127
160
  }
@@ -241,6 +274,9 @@ export async function handleGenerateVideo(request, apiClient, progress) {
241
274
  return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
242
275
  }
243
276
  const model = args.model || process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL || FALLBACK_MODEL;
277
+ // Sora deprecation warning — OpenAI is removing the Videos API and all
278
+ // Sora 2 model aliases on September 24, 2026. Warn and suggest alternatives.
279
+ const deprecationWarning = checkSoraDeprecation(model);
244
280
  // Audit entry — video is the most expensive tool we have. Always log
245
281
  // model, resolution, duration, and a safe prompt preview so unintended
246
282
  // spend can be traced.
@@ -301,13 +337,16 @@ export async function handleGenerateVideo(request, apiClient, progress) {
301
337
  });
302
338
  }
303
339
  if (outcome.kind === 'timeout') {
340
+ const timeoutContent = [];
341
+ if (deprecationWarning) {
342
+ timeoutContent.push({ type: 'text', text: deprecationWarning });
343
+ }
344
+ timeoutContent.push({
345
+ type: 'text',
346
+ text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
347
+ });
304
348
  return {
305
- content: [
306
- {
307
- type: 'text',
308
- text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
309
- },
310
- ],
349
+ content: timeoutContent,
311
350
  isError: false,
312
351
  _meta: {
313
352
  server_version: SERVER_VERSION,
@@ -320,6 +359,11 @@ export async function handleGenerateVideo(request, apiClient, progress) {
320
359
  }
321
360
  try {
322
361
  const { content, _meta } = await finalizeCompletedJob(apiClient, outcome.status, safeSavePath);
362
+ // Prepend deprecation warning if applicable
363
+ if (deprecationWarning) {
364
+ content.unshift({ type: 'text', text: deprecationWarning });
365
+ _meta.deprecated_model = true;
366
+ }
323
367
  return { content, _meta };
324
368
  }
325
369
  catch (err) {
@@ -14,11 +14,31 @@
14
14
  */
15
15
  import path from 'node:path';
16
16
  import { promises as fs } from 'node:fs';
17
+ import os from 'node:os';
18
+ /**
19
+ * Resolve the output root directory. On Windows, `process.cwd()` can be
20
+ * a system directory (e.g. `C:\Windows\System32`) when spawned by MCP
21
+ * clients without a working directory override. We detect that case and
22
+ * fall back to a writable temp directory to avoid EPERM errors.
23
+ */
17
24
  function getOutputRoot() {
18
25
  const override = process.env.OPENROUTER_OUTPUT_DIR;
19
26
  if (override && override.length > 0)
20
27
  return path.resolve(override);
21
- return process.cwd();
28
+ const cwd = process.cwd();
29
+ // On Windows, avoid using system directories as the default output root.
30
+ // Common non-writable defaults when MCP clients spawn without a cwd:
31
+ // C:\Windows\System32, C:\Windows, C:\Program Files\...
32
+ if (process.platform === 'win32') {
33
+ const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
34
+ if (cwdLower.startsWith('c:/windows') ||
35
+ cwdLower.startsWith('c:/program files') ||
36
+ cwdLower.startsWith('c:/program files (x86)')) {
37
+ const fallback = path.join(os.homedir(), 'openrouter-mcp-output');
38
+ return fallback;
39
+ }
40
+ }
41
+ return cwd;
22
42
  }
23
43
  function isUnsafeMode() {
24
44
  const v = process.env.OPENROUTER_ALLOW_UNSAFE_PATHS;
@@ -89,9 +109,7 @@ async function findExistingAncestor(dir) {
89
109
  /**
90
110
  * Root-resolution for caller-supplied INPUT paths. Prefers
91
111
  * `OPENROUTER_INPUT_DIR`, then `OPENROUTER_OUTPUT_DIR`, then `process.cwd()`.
92
- * This mirrors the semantics `generate_image`'s `input_images` originally
93
- * shipped with; exposing it here lets `generate_video`'s frame and
94
- * reference images use the same sandbox.
112
+ * On Windows, applies the same system-directory detection as getOutputRoot.
95
113
  */
96
114
  function getInputRoot() {
97
115
  const inputDir = process.env.OPENROUTER_INPUT_DIR;
@@ -100,7 +118,16 @@ function getInputRoot() {
100
118
  const outputDir = process.env.OPENROUTER_OUTPUT_DIR;
101
119
  if (outputDir && outputDir.length > 0)
102
120
  return path.resolve(outputDir);
103
- return process.cwd();
121
+ const cwd = process.cwd();
122
+ if (process.platform === 'win32') {
123
+ const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
124
+ if (cwdLower.startsWith('c:/windows') ||
125
+ cwdLower.startsWith('c:/program files') ||
126
+ cwdLower.startsWith('c:/program files (x86)')) {
127
+ return path.join(os.homedir(), 'openrouter-mcp-output');
128
+ }
129
+ }
130
+ return cwd;
104
131
  }
105
132
  /**
106
133
  * Resolve and validate a caller-supplied INPUT path. Unlike
@@ -0,0 +1,20 @@
1
+ import type { OpenRouterAPIClient } from '../openrouter-api.js';
2
+ import { type CacheOptions } from './cache.js';
3
+ export interface SpeechToTextRequest extends CacheOptions {
4
+ audio_path: string;
5
+ model?: string;
6
+ language?: string;
7
+ response_format?: string;
8
+ temperature?: number;
9
+ }
10
+ export declare function handleSpeechToText(request: {
11
+ params: {
12
+ arguments: SpeechToTextRequest;
13
+ };
14
+ }, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
15
+ content: {
16
+ type: "text";
17
+ text: string;
18
+ }[];
19
+ _meta: Record<string, unknown>;
20
+ }>;
@@ -0,0 +1,140 @@
1
+ /**
2
+ * speech_to_text — uses OpenRouter's dedicated POST /api/v1/audio/transcriptions
3
+ * endpoint (launched May 2026) for speech-to-text transcription. Faster and more
4
+ * cost-efficient than routing through chat completions for pure transcription.
5
+ *
6
+ * Supported models: OpenAI Whisper-1, GPT-4o Transcribe, GPT-4o Mini Transcribe,
7
+ * Mistral Voxtral Mini Transcribe.
8
+ */
9
+ import { promises as fs } from 'fs';
10
+ import path from 'node:path';
11
+ import { resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
12
+ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
13
+ import { SERVER_VERSION } from '../version.js';
14
+ import { logger } from '../logger.js';
15
+ import { classifyUpstreamError } from './openrouter-errors.js';
16
+ import { buildCacheHeaders } from './cache.js';
17
+ const DEFAULT_MODEL = 'openai/whisper-1';
18
+ const VALID_RESPONSE_FORMATS = new Set(['json', 'text', 'srt', 'verbose_json', 'vtt']);
19
+ /** Infer audio format from file extension. */
20
+ function audioFormatFromExt(ext) {
21
+ const normalized = ext.toLowerCase().replace('.', '');
22
+ switch (normalized) {
23
+ case 'mp3': return 'mp3';
24
+ case 'mp4':
25
+ case 'm4a': return 'mp4';
26
+ case 'wav': return 'wav';
27
+ case 'flac': return 'flac';
28
+ case 'ogg':
29
+ case 'oga': return 'ogg';
30
+ case 'webm': return 'webm';
31
+ case 'opus': return 'opus';
32
+ default: return 'mp3';
33
+ }
34
+ }
35
+ /**
36
+ * Resolve audio input to base64 + format, supporting:
37
+ * - data: URLs (pass through)
38
+ * - http(s) URLs (fetch)
39
+ * - local file paths (sandboxed read)
40
+ */
41
+ async function resolveAudioInput(audioPath) {
42
+ const trimmed = audioPath.trim();
43
+ if (!trimmed)
44
+ throw new Error('audio_path is empty');
45
+ // Data URL
46
+ if (trimmed.startsWith('data:')) {
47
+ const match = trimmed.match(/^data:audio\/([^;,]+)(?:;[^,]*)*;base64,(.+)$/);
48
+ if (!match)
49
+ throw new Error('Invalid audio data URL format');
50
+ return { data: match[2], format: match[1] };
51
+ }
52
+ // HTTP URL
53
+ if (/^https?:\/\//i.test(trimmed)) {
54
+ const { fetchHttpResource } = await import('./fetch-utils.js');
55
+ const { buffer, contentType } = await fetchHttpResource(trimmed, {
56
+ timeoutMs: 60_000,
57
+ maxBytes: 100 * 1024 * 1024,
58
+ maxRedirects: 8,
59
+ });
60
+ const format = contentType?.match(/audio\/(\w+)/)?.[1] || 'mp3';
61
+ return { data: buffer.toString('base64'), format };
62
+ }
63
+ // Local file
64
+ const abs = await resolveSafeInputPath(trimmed);
65
+ const buf = await fs.readFile(abs);
66
+ const ext = path.extname(abs);
67
+ const format = audioFormatFromExt(ext);
68
+ return { data: buf.toString('base64'), format };
69
+ }
70
+ export async function handleSpeechToText(request, apiClient) {
71
+ const args = request.params.arguments ?? {};
72
+ const { audio_path, model, language, response_format, temperature, cache, cache_ttl, cache_clear, } = args;
73
+ if (!audio_path?.trim()) {
74
+ return toolError(ErrorCode.INVALID_INPUT, 'audio_path is required.');
75
+ }
76
+ if (response_format && !VALID_RESPONSE_FORMATS.has(response_format)) {
77
+ return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_RESPONSE_FORMATS].join(', ')}.`);
78
+ }
79
+ logger.audit('speech_to_text.start', {
80
+ model: model || DEFAULT_MODEL,
81
+ audio_path: audio_path.startsWith('data:') ? 'data_url' : audio_path.slice(0, 80),
82
+ language,
83
+ response_format,
84
+ });
85
+ // Resolve audio input
86
+ let audioInput;
87
+ try {
88
+ audioInput = await resolveAudioInput(audio_path);
89
+ }
90
+ catch (err) {
91
+ if (err instanceof UnsafeOutputPathError)
92
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
93
+ const msg = err instanceof Error ? err.message : String(err);
94
+ if (msg.includes('Blocked host'))
95
+ return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
96
+ return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
97
+ }
98
+ // Build request body
99
+ const body = {
100
+ model: model || DEFAULT_MODEL,
101
+ input_audio: {
102
+ data: audioInput.data,
103
+ format: audioInput.format,
104
+ },
105
+ };
106
+ if (language)
107
+ body.language = language;
108
+ if (response_format)
109
+ body.response_format = response_format;
110
+ if (typeof temperature === 'number')
111
+ body.temperature = temperature;
112
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
113
+ let response;
114
+ try {
115
+ response = await apiClient.transcribeAudio(body, headers);
116
+ }
117
+ catch (err) {
118
+ return classifyUpstreamError(err, 'speech_to_text');
119
+ }
120
+ const text = response.text;
121
+ if (!text) {
122
+ return toolError(ErrorCode.INTERNAL, 'Transcription returned no text.', {
123
+ response_keys: Object.keys(response),
124
+ });
125
+ }
126
+ const baseMeta = {
127
+ server_version: SERVER_VERSION,
128
+ model: model || DEFAULT_MODEL,
129
+ };
130
+ if (response.language)
131
+ baseMeta.language = response.language;
132
+ if (response.duration)
133
+ baseMeta.duration_seconds = response.duration;
134
+ if (response.usage)
135
+ baseMeta.usage = response.usage;
136
+ return {
137
+ content: [{ type: 'text', text }],
138
+ _meta: baseMeta,
139
+ };
140
+ }
@@ -0,0 +1,29 @@
1
+ import type { OpenRouterAPIClient } from '../openrouter-api.js';
2
+ import { type CacheOptions } from './cache.js';
3
+ export interface TextToSpeechRequest extends CacheOptions {
4
+ input: string;
5
+ model?: string;
6
+ voice?: string;
7
+ response_format?: string;
8
+ speed?: number;
9
+ instructions?: string;
10
+ save_path?: string;
11
+ }
12
+ export declare function handleTextToSpeech(request: {
13
+ params: {
14
+ arguments: TextToSpeechRequest;
15
+ };
16
+ }, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
17
+ content: ({
18
+ type: "text";
19
+ text: string;
20
+ mimeType?: undefined;
21
+ data?: undefined;
22
+ } | {
23
+ type: "audio";
24
+ mimeType: string;
25
+ data: string;
26
+ text?: undefined;
27
+ })[];
28
+ _meta: Record<string, unknown>;
29
+ }>;
@@ -0,0 +1,105 @@
1
+ /**
2
+ * text_to_speech — uses OpenRouter's dedicated POST /api/v1/audio/speech
3
+ * endpoint (launched May 2026) for text-to-speech. Faster and more cost-efficient
4
+ * than routing through chat completions with audio modality.
5
+ *
6
+ * Supported providers: OpenAI (GPT-4o Mini TTS), Google (Gemini Flash TTS),
7
+ * Mistral (Voxtral Mini TTS).
8
+ */
9
+ import { promises as fs } from 'fs';
10
+ import { extname } from 'path';
11
+ import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
12
+ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
13
+ import { SERVER_VERSION } from '../version.js';
14
+ import { logger } from '../logger.js';
15
+ import { classifyUpstreamError } from './openrouter-errors.js';
16
+ import { buildCacheHeaders } from './cache.js';
17
+ const DEFAULT_MODEL = 'openai/gpt-4o-mini-tts-2025-12-15';
18
+ const DEFAULT_VOICE = 'alloy';
19
+ const VALID_FORMATS = new Set(['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm']);
20
+ export async function handleTextToSpeech(request, apiClient) {
21
+ const args = request.params.arguments ?? {};
22
+ const { input, model, voice, response_format, speed, instructions, save_path, cache, cache_ttl, cache_clear, } = args;
23
+ if (!input?.trim()) {
24
+ return toolError(ErrorCode.INVALID_INPUT, 'input text is required.');
25
+ }
26
+ if (response_format && !VALID_FORMATS.has(response_format)) {
27
+ return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_FORMATS].join(', ')}.`);
28
+ }
29
+ logger.audit('text_to_speech.start', {
30
+ model: model || DEFAULT_MODEL,
31
+ voice: voice || DEFAULT_VOICE,
32
+ response_format: response_format || 'mp3',
33
+ input_preview: input.slice(0, 80),
34
+ save_path: save_path ? 'provided' : 'none',
35
+ });
36
+ // Resolve save path early
37
+ let safeSavePath = null;
38
+ if (save_path) {
39
+ try {
40
+ safeSavePath = await resolveSafeOutputPath(save_path);
41
+ }
42
+ catch (err) {
43
+ if (err instanceof UnsafeOutputPathError)
44
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
45
+ return toolErrorFrom(ErrorCode.INTERNAL, err);
46
+ }
47
+ }
48
+ // Build request body
49
+ const body = {
50
+ model: model || DEFAULT_MODEL,
51
+ input,
52
+ voice: voice || DEFAULT_VOICE,
53
+ };
54
+ if (response_format)
55
+ body.response_format = response_format;
56
+ if (typeof speed === 'number' && speed > 0)
57
+ body.speed = speed;
58
+ if (instructions)
59
+ body.instructions = instructions;
60
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
61
+ let result;
62
+ try {
63
+ result = await apiClient.generateSpeech(body, headers);
64
+ }
65
+ catch (err) {
66
+ return classifyUpstreamError(err, 'text_to_speech');
67
+ }
68
+ const { buffer, contentType } = result;
69
+ const mimeType = contentType.split(';')[0]?.trim() || 'audio/mpeg';
70
+ // Determine file extension from format
71
+ const ext = response_format || 'mp3';
72
+ const baseMeta = {
73
+ server_version: SERVER_VERSION,
74
+ model: model || DEFAULT_MODEL,
75
+ mime: mimeType,
76
+ size_bytes: buffer.length,
77
+ voice: voice || DEFAULT_VOICE,
78
+ };
79
+ if (safeSavePath) {
80
+ // Ensure extension matches
81
+ const currentExt = extname(safeSavePath).toLowerCase().slice(1);
82
+ const actualPath = currentExt === ext ? safeSavePath : `${safeSavePath}.${ext}`;
83
+ try {
84
+ await fs.writeFile(actualPath, buffer);
85
+ }
86
+ catch (err) {
87
+ return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
88
+ }
89
+ baseMeta.save_path = actualPath;
90
+ return {
91
+ content: [
92
+ { type: 'text', text: `Speech saved to: ${actualPath}` },
93
+ { type: 'audio', mimeType, data: buffer.toString('base64') },
94
+ ],
95
+ _meta: baseMeta,
96
+ };
97
+ }
98
+ return {
99
+ content: [
100
+ { type: 'text', text: `Speech generated (${buffer.length} bytes, ${mimeType}).` },
101
+ { type: 'audio', mimeType, data: buffer.toString('base64') },
102
+ ],
103
+ _meta: baseMeta,
104
+ };
105
+ }