@stabgan/openrouter-mcp-multimodal 4.5.3 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/openrouter-api.d.ts +45 -0
- package/dist/openrouter-api.js +50 -0
- package/dist/tool-descriptions.d.ts +1 -1
- package/dist/tool-descriptions.js +152 -2
- package/dist/tool-handlers/async-chat.d.ts +51 -0
- package/dist/tool-handlers/async-chat.js +216 -0
- package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
- package/dist/tool-handlers/generate-image-dedicated.js +176 -0
- package/dist/tool-handlers/generate-video.d.ts +2 -2
- package/dist/tool-handlers/generate-video.js +53 -9
- package/dist/tool-handlers/path-safety.js +32 -5
- package/dist/tool-handlers/speech-to-text.d.ts +20 -0
- package/dist/tool-handlers/speech-to-text.js +140 -0
- package/dist/tool-handlers/text-to-speech.d.ts +29 -0
- package/dist/tool-handlers/text-to-speech.js +105 -0
- package/dist/tool-handlers.js +228 -2
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +2 -2
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* generate_image_dedicated — uses OpenRouter's dedicated POST /api/v1/images
|
|
3
|
+
* endpoint (launched June 2026) for image generation. Supports normalized
|
|
4
|
+
* resolution tiers, aspect ratios, quality levels, output formats, and
|
|
5
|
+
* input_references for image-to-image workflows.
|
|
6
|
+
*
|
|
7
|
+
* This is distinct from the original `generate_image` tool which uses chat
|
|
8
|
+
* completions with `modalities: ['image', 'text']`. New image models are
|
|
9
|
+
* added exclusively to this dedicated endpoint.
|
|
10
|
+
*/
|
|
11
|
+
import { promises as fs } from 'fs';
|
|
12
|
+
import path from 'node:path';
|
|
13
|
+
import { resolveSafeOutputPath, resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
14
|
+
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
15
|
+
import { SERVER_VERSION } from '../version.js';
|
|
16
|
+
import { logger } from '../logger.js';
|
|
17
|
+
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
18
|
+
import { buildCacheHeaders } from './cache.js';
|
|
19
|
+
const DEFAULT_MODEL = 'google/gemini-2.5-flash-image';
|
|
20
|
+
const VALID_RESOLUTIONS = new Set(['512', '0.5K', '1K', '2K', '4K']);
|
|
21
|
+
const VALID_QUALITIES = new Set(['auto', 'low', 'medium', 'high']);
|
|
22
|
+
const VALID_OUTPUT_FORMATS = new Set(['png', 'jpeg', 'webp', 'svg']);
|
|
23
|
+
/**
|
|
24
|
+
* Resolve an input image reference (local path, URL, or data URL) into the
|
|
25
|
+
* OpenRouter `input_references` shape: `{ type: "image_url", image_url: { url } }`.
|
|
26
|
+
*/
|
|
27
|
+
async function resolveReference(source) {
|
|
28
|
+
const trimmed = source.trim();
|
|
29
|
+
if (!trimmed)
|
|
30
|
+
throw new Error('Empty input_references entry');
|
|
31
|
+
// Data URLs and HTTP URLs pass through directly
|
|
32
|
+
if (trimmed.startsWith('data:') || /^https?:\/\//i.test(trimmed)) {
|
|
33
|
+
return { type: 'image_url', image_url: { url: trimmed } };
|
|
34
|
+
}
|
|
35
|
+
// Local file: sandbox, read, and convert to data URL
|
|
36
|
+
const abs = await resolveSafeInputPath(trimmed);
|
|
37
|
+
const buf = await fs.readFile(abs);
|
|
38
|
+
const ext = path.extname(abs).toLowerCase();
|
|
39
|
+
const mime = ext === '.png' ? 'image/png' :
|
|
40
|
+
ext === '.webp' ? 'image/webp' :
|
|
41
|
+
ext === '.gif' ? 'image/gif' :
|
|
42
|
+
ext === '.svg' ? 'image/svg+xml' :
|
|
43
|
+
'image/jpeg';
|
|
44
|
+
const dataUrl = `data:${mime};base64,${buf.toString('base64')}`;
|
|
45
|
+
return { type: 'image_url', image_url: { url: dataUrl } };
|
|
46
|
+
}
|
|
47
|
+
export async function handleGenerateImageDedicated(request, apiClient) {
|
|
48
|
+
const args = request.params.arguments ?? {};
|
|
49
|
+
const { prompt, model, resolution, aspect_ratio, quality, output_format, n, input_references, save_path, provider, cache, cache_ttl, cache_clear, } = args;
|
|
50
|
+
if (!prompt?.trim()) {
|
|
51
|
+
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
52
|
+
}
|
|
53
|
+
logger.audit('generate_image_dedicated.start', {
|
|
54
|
+
model: model || DEFAULT_MODEL,
|
|
55
|
+
prompt_preview: prompt.slice(0, 80),
|
|
56
|
+
resolution,
|
|
57
|
+
aspect_ratio,
|
|
58
|
+
quality,
|
|
59
|
+
output_format,
|
|
60
|
+
input_references_count: input_references?.length ?? 0,
|
|
61
|
+
save_path: save_path ? 'provided' : 'none',
|
|
62
|
+
});
|
|
63
|
+
// Validate enums
|
|
64
|
+
if (resolution && !VALID_RESOLUTIONS.has(resolution)) {
|
|
65
|
+
return toolError(ErrorCode.INVALID_INPUT, `resolution '${resolution}' is not supported. Valid: ${[...VALID_RESOLUTIONS].join(', ')}.`);
|
|
66
|
+
}
|
|
67
|
+
if (quality && !VALID_QUALITIES.has(quality)) {
|
|
68
|
+
return toolError(ErrorCode.INVALID_INPUT, `quality '${quality}' is not supported. Valid: ${[...VALID_QUALITIES].join(', ')}.`);
|
|
69
|
+
}
|
|
70
|
+
if (output_format && !VALID_OUTPUT_FORMATS.has(output_format)) {
|
|
71
|
+
return toolError(ErrorCode.INVALID_INPUT, `output_format '${output_format}' is not supported. Valid: ${[...VALID_OUTPUT_FORMATS].join(', ')}.`);
|
|
72
|
+
}
|
|
73
|
+
// Resolve save path early
|
|
74
|
+
let safeSavePath = null;
|
|
75
|
+
if (save_path) {
|
|
76
|
+
try {
|
|
77
|
+
safeSavePath = await resolveSafeOutputPath(save_path);
|
|
78
|
+
}
|
|
79
|
+
catch (err) {
|
|
80
|
+
if (err instanceof UnsafeOutputPathError)
|
|
81
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
82
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
// Build request body
|
|
86
|
+
const body = {
|
|
87
|
+
model: model || DEFAULT_MODEL,
|
|
88
|
+
prompt,
|
|
89
|
+
};
|
|
90
|
+
if (resolution)
|
|
91
|
+
body.resolution = resolution;
|
|
92
|
+
if (aspect_ratio)
|
|
93
|
+
body.aspect_ratio = aspect_ratio;
|
|
94
|
+
if (quality)
|
|
95
|
+
body.quality = quality;
|
|
96
|
+
if (output_format)
|
|
97
|
+
body.output_format = output_format;
|
|
98
|
+
if (typeof n === 'number' && n > 0)
|
|
99
|
+
body.n = n;
|
|
100
|
+
if (provider && typeof provider === 'object')
|
|
101
|
+
body.provider = provider;
|
|
102
|
+
// Resolve input references
|
|
103
|
+
if (input_references?.length) {
|
|
104
|
+
try {
|
|
105
|
+
const refs = await Promise.all(input_references.map(resolveReference));
|
|
106
|
+
body.input_references = refs;
|
|
107
|
+
}
|
|
108
|
+
catch (err) {
|
|
109
|
+
if (err instanceof UnsafeOutputPathError)
|
|
110
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
111
|
+
return toolErrorFrom(ErrorCode.INVALID_INPUT, err, 'input_references');
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
// Build cache headers
|
|
115
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
116
|
+
let response;
|
|
117
|
+
try {
|
|
118
|
+
response = await apiClient.generateImage(body, headers);
|
|
119
|
+
}
|
|
120
|
+
catch (err) {
|
|
121
|
+
return classifyUpstreamError(err, 'generate_image_dedicated');
|
|
122
|
+
}
|
|
123
|
+
const images = response.data ?? [];
|
|
124
|
+
if (!images.length || (!images[0]?.b64_json && !images[0]?.url)) {
|
|
125
|
+
return toolError(ErrorCode.UPSTREAM_REFUSED, 'Model returned no image data.', {
|
|
126
|
+
response_keys: Object.keys(response),
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
const firstImage = images[0];
|
|
130
|
+
const imageData = firstImage.b64_json;
|
|
131
|
+
const mimeType = output_format === 'png' ? 'image/png' :
|
|
132
|
+
output_format === 'webp' ? 'image/webp' :
|
|
133
|
+
output_format === 'svg' ? 'image/svg+xml' :
|
|
134
|
+
output_format === 'jpeg' ? 'image/jpeg' :
|
|
135
|
+
'image/png'; // default
|
|
136
|
+
const baseMeta = {
|
|
137
|
+
server_version: SERVER_VERSION,
|
|
138
|
+
model: model || DEFAULT_MODEL,
|
|
139
|
+
images_count: images.length,
|
|
140
|
+
};
|
|
141
|
+
if (response.usage)
|
|
142
|
+
baseMeta.usage = response.usage;
|
|
143
|
+
if (firstImage.revised_prompt)
|
|
144
|
+
baseMeta.revised_prompt = firstImage.revised_prompt;
|
|
145
|
+
// Save to file if requested
|
|
146
|
+
if (safeSavePath && imageData) {
|
|
147
|
+
try {
|
|
148
|
+
await fs.writeFile(safeSavePath, imageData, { encoding: 'base64' });
|
|
149
|
+
}
|
|
150
|
+
catch (err) {
|
|
151
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
|
|
152
|
+
}
|
|
153
|
+
baseMeta.save_path = safeSavePath;
|
|
154
|
+
return {
|
|
155
|
+
content: [
|
|
156
|
+
{ type: 'text', text: `Image saved to: ${safeSavePath}` },
|
|
157
|
+
...(imageData ? [{ type: 'image', mimeType, data: imageData }] : []),
|
|
158
|
+
],
|
|
159
|
+
_meta: baseMeta,
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
// Return inline
|
|
163
|
+
if (imageData) {
|
|
164
|
+
return {
|
|
165
|
+
content: [{ type: 'image', mimeType, data: imageData }],
|
|
166
|
+
_meta: baseMeta,
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
// URL-only response (some models return URLs instead of base64)
|
|
170
|
+
return {
|
|
171
|
+
content: [
|
|
172
|
+
{ type: 'text', text: `Image generated. URL: ${firstImage.url}` },
|
|
173
|
+
],
|
|
174
|
+
_meta: { ...baseMeta, image_url: firstImage.url },
|
|
175
|
+
};
|
|
176
|
+
}
|
|
@@ -34,7 +34,7 @@ export declare function handleGenerateVideo(request: {
|
|
|
34
34
|
};
|
|
35
35
|
}, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
|
|
36
36
|
content: {
|
|
37
|
-
type:
|
|
37
|
+
type: string;
|
|
38
38
|
text: string;
|
|
39
39
|
}[];
|
|
40
40
|
isError: false;
|
|
@@ -97,7 +97,7 @@ export declare function handleGenerateVideoFromImage(request: {
|
|
|
97
97
|
};
|
|
98
98
|
}, apiClient: OpenRouterAPIClient, progress?: ProgressHook): Promise<import("../errors.js").ToolErrorResult | {
|
|
99
99
|
content: {
|
|
100
|
-
type:
|
|
100
|
+
type: string;
|
|
101
101
|
text: string;
|
|
102
102
|
}[];
|
|
103
103
|
isError: false;
|
|
@@ -11,6 +11,34 @@ const DEFAULT_POLL_INTERVAL_MS = 15_000;
|
|
|
11
11
|
const DEFAULT_MAX_WAIT_MS = 10 * 60_000;
|
|
12
12
|
const MIN_POLL_INTERVAL_MS = 50; // just to avoid a 0ms busy-loop if a caller omits
|
|
13
13
|
const INLINE_RETURN_CEILING_BYTES = 10 * 1024 * 1024;
|
|
14
|
+
/** Models deprecated by OpenAI — removal date: 2026-09-24. */
|
|
15
|
+
const SORA_DEPRECATED_MODELS = new Set([
|
|
16
|
+
'openai/sora-2',
|
|
17
|
+
'openai/sora-2-pro',
|
|
18
|
+
'openai/sora-2-2025-10-06',
|
|
19
|
+
'openai/sora-2-2025-12-08',
|
|
20
|
+
'openai/sora-2-pro-2025-10-06',
|
|
21
|
+
]);
|
|
22
|
+
const SORA_ALTERNATIVES = [
|
|
23
|
+
'google/veo-3.1 (recommended — fast, audio support)',
|
|
24
|
+
'google/veo-3.1-fast (budget-friendly)',
|
|
25
|
+
'bytedance/seedance-2.0 (high quality)',
|
|
26
|
+
'bytedance/seedance-2.0-fast (fast turnaround)',
|
|
27
|
+
'alibaba/wan-2.7 (good for artistic styles)',
|
|
28
|
+
];
|
|
29
|
+
/**
|
|
30
|
+
* Check if the model is a deprecated Sora model and return a warning string,
|
|
31
|
+
* or null if no deprecation applies.
|
|
32
|
+
*/
|
|
33
|
+
function checkSoraDeprecation(model) {
|
|
34
|
+
const normalized = model.toLowerCase().trim();
|
|
35
|
+
if (!SORA_DEPRECATED_MODELS.has(normalized) && !normalized.startsWith('openai/sora')) {
|
|
36
|
+
return null;
|
|
37
|
+
}
|
|
38
|
+
return (`⚠️ DEPRECATION WARNING: ${model} is deprecated by OpenAI and will be removed from the API on September 24, 2026. ` +
|
|
39
|
+
`Your request will still be attempted, but may fail. Recommended alternatives:\n` +
|
|
40
|
+
SORA_ALTERNATIVES.map((a) => ` • ${a}`).join('\n'));
|
|
41
|
+
}
|
|
14
42
|
function getMaxInlineBytes() {
|
|
15
43
|
return readEnvInt('OPENROUTER_VIDEO_INLINE_MAX_BYTES', INLINE_RETURN_CEILING_BYTES, 4096);
|
|
16
44
|
}
|
|
@@ -94,8 +122,9 @@ async function attachFrameImages(args, body) {
|
|
|
94
122
|
? {
|
|
95
123
|
kind: 'frame',
|
|
96
124
|
entry: {
|
|
125
|
+
type: 'image_url',
|
|
126
|
+
image_url: { url: `data:${img.mime};base64,${img.data}` },
|
|
97
127
|
frame_type: 'first_frame',
|
|
98
|
-
image: { url: `data:${img.mime};base64,${img.data}` },
|
|
99
128
|
},
|
|
100
129
|
}
|
|
101
130
|
: null));
|
|
@@ -105,8 +134,9 @@ async function attachFrameImages(args, body) {
|
|
|
105
134
|
? {
|
|
106
135
|
kind: 'frame',
|
|
107
136
|
entry: {
|
|
137
|
+
type: 'image_url',
|
|
138
|
+
image_url: { url: `data:${img.mime};base64,${img.data}` },
|
|
108
139
|
frame_type: 'last_frame',
|
|
109
|
-
image: { url: `data:${img.mime};base64,${img.data}` },
|
|
110
140
|
},
|
|
111
141
|
}
|
|
112
142
|
: null));
|
|
@@ -121,7 +151,10 @@ async function attachFrameImages(args, body) {
|
|
|
121
151
|
const refResults = await Promise.all(args.reference_images.map((src) => prepareImageInput(src)));
|
|
122
152
|
const refs = refResults
|
|
123
153
|
.filter((img) => img !== null)
|
|
124
|
-
.map((img) => ({
|
|
154
|
+
.map((img) => ({
|
|
155
|
+
type: 'image_url',
|
|
156
|
+
image_url: { url: `data:${img.mime};base64,${img.data}` },
|
|
157
|
+
}));
|
|
125
158
|
if (refs.length)
|
|
126
159
|
body.input_references = refs;
|
|
127
160
|
}
|
|
@@ -241,6 +274,9 @@ export async function handleGenerateVideo(request, apiClient, progress) {
|
|
|
241
274
|
return toolError(ErrorCode.INVALID_INPUT, 'prompt is required.');
|
|
242
275
|
}
|
|
243
276
|
const model = args.model || process.env.OPENROUTER_DEFAULT_VIDEO_GEN_MODEL || FALLBACK_MODEL;
|
|
277
|
+
// Sora deprecation warning — OpenAI is removing the Videos API and all
|
|
278
|
+
// Sora 2 model aliases on September 24, 2026. Warn and suggest alternatives.
|
|
279
|
+
const deprecationWarning = checkSoraDeprecation(model);
|
|
244
280
|
// Audit entry — video is the most expensive tool we have. Always log
|
|
245
281
|
// model, resolution, duration, and a safe prompt preview so unintended
|
|
246
282
|
// spend can be traced.
|
|
@@ -301,13 +337,16 @@ export async function handleGenerateVideo(request, apiClient, progress) {
|
|
|
301
337
|
});
|
|
302
338
|
}
|
|
303
339
|
if (outcome.kind === 'timeout') {
|
|
340
|
+
const timeoutContent = [];
|
|
341
|
+
if (deprecationWarning) {
|
|
342
|
+
timeoutContent.push({ type: 'text', text: deprecationWarning });
|
|
343
|
+
}
|
|
344
|
+
timeoutContent.push({
|
|
345
|
+
type: 'text',
|
|
346
|
+
text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
|
|
347
|
+
});
|
|
304
348
|
return {
|
|
305
|
-
content:
|
|
306
|
-
{
|
|
307
|
-
type: 'text',
|
|
308
|
-
text: `Video still generating after ${maxWaitMs}ms. Use get_video_status with video_id=${envelope.id} to resume.`,
|
|
309
|
-
},
|
|
310
|
-
],
|
|
349
|
+
content: timeoutContent,
|
|
311
350
|
isError: false,
|
|
312
351
|
_meta: {
|
|
313
352
|
server_version: SERVER_VERSION,
|
|
@@ -320,6 +359,11 @@ export async function handleGenerateVideo(request, apiClient, progress) {
|
|
|
320
359
|
}
|
|
321
360
|
try {
|
|
322
361
|
const { content, _meta } = await finalizeCompletedJob(apiClient, outcome.status, safeSavePath);
|
|
362
|
+
// Prepend deprecation warning if applicable
|
|
363
|
+
if (deprecationWarning) {
|
|
364
|
+
content.unshift({ type: 'text', text: deprecationWarning });
|
|
365
|
+
_meta.deprecated_model = true;
|
|
366
|
+
}
|
|
323
367
|
return { content, _meta };
|
|
324
368
|
}
|
|
325
369
|
catch (err) {
|
|
@@ -14,11 +14,31 @@
|
|
|
14
14
|
*/
|
|
15
15
|
import path from 'node:path';
|
|
16
16
|
import { promises as fs } from 'node:fs';
|
|
17
|
+
import os from 'node:os';
|
|
18
|
+
/**
|
|
19
|
+
* Resolve the output root directory. On Windows, `process.cwd()` can be
|
|
20
|
+
* a system directory (e.g. `C:\Windows\System32`) when spawned by MCP
|
|
21
|
+
* clients without a working directory override. We detect that case and
|
|
22
|
+
* fall back to a writable temp directory to avoid EPERM errors.
|
|
23
|
+
*/
|
|
17
24
|
function getOutputRoot() {
|
|
18
25
|
const override = process.env.OPENROUTER_OUTPUT_DIR;
|
|
19
26
|
if (override && override.length > 0)
|
|
20
27
|
return path.resolve(override);
|
|
21
|
-
|
|
28
|
+
const cwd = process.cwd();
|
|
29
|
+
// On Windows, avoid using system directories as the default output root.
|
|
30
|
+
// Common non-writable defaults when MCP clients spawn without a cwd:
|
|
31
|
+
// C:\Windows\System32, C:\Windows, C:\Program Files\...
|
|
32
|
+
if (process.platform === 'win32') {
|
|
33
|
+
const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
|
|
34
|
+
if (cwdLower.startsWith('c:/windows') ||
|
|
35
|
+
cwdLower.startsWith('c:/program files') ||
|
|
36
|
+
cwdLower.startsWith('c:/program files (x86)')) {
|
|
37
|
+
const fallback = path.join(os.homedir(), 'openrouter-mcp-output');
|
|
38
|
+
return fallback;
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
return cwd;
|
|
22
42
|
}
|
|
23
43
|
function isUnsafeMode() {
|
|
24
44
|
const v = process.env.OPENROUTER_ALLOW_UNSAFE_PATHS;
|
|
@@ -89,9 +109,7 @@ async function findExistingAncestor(dir) {
|
|
|
89
109
|
/**
|
|
90
110
|
* Root-resolution for caller-supplied INPUT paths. Prefers
|
|
91
111
|
* `OPENROUTER_INPUT_DIR`, then `OPENROUTER_OUTPUT_DIR`, then `process.cwd()`.
|
|
92
|
-
*
|
|
93
|
-
* shipped with; exposing it here lets `generate_video`'s frame and
|
|
94
|
-
* reference images use the same sandbox.
|
|
112
|
+
* On Windows, applies the same system-directory detection as getOutputRoot.
|
|
95
113
|
*/
|
|
96
114
|
function getInputRoot() {
|
|
97
115
|
const inputDir = process.env.OPENROUTER_INPUT_DIR;
|
|
@@ -100,7 +118,16 @@ function getInputRoot() {
|
|
|
100
118
|
const outputDir = process.env.OPENROUTER_OUTPUT_DIR;
|
|
101
119
|
if (outputDir && outputDir.length > 0)
|
|
102
120
|
return path.resolve(outputDir);
|
|
103
|
-
|
|
121
|
+
const cwd = process.cwd();
|
|
122
|
+
if (process.platform === 'win32') {
|
|
123
|
+
const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
|
|
124
|
+
if (cwdLower.startsWith('c:/windows') ||
|
|
125
|
+
cwdLower.startsWith('c:/program files') ||
|
|
126
|
+
cwdLower.startsWith('c:/program files (x86)')) {
|
|
127
|
+
return path.join(os.homedir(), 'openrouter-mcp-output');
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
return cwd;
|
|
104
131
|
}
|
|
105
132
|
/**
|
|
106
133
|
* Resolve and validate a caller-supplied INPUT path. Unlike
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
import type { OpenRouterAPIClient } from '../openrouter-api.js';
|
|
2
|
+
import { type CacheOptions } from './cache.js';
|
|
3
|
+
export interface SpeechToTextRequest extends CacheOptions {
|
|
4
|
+
audio_path: string;
|
|
5
|
+
model?: string;
|
|
6
|
+
language?: string;
|
|
7
|
+
response_format?: string;
|
|
8
|
+
temperature?: number;
|
|
9
|
+
}
|
|
10
|
+
export declare function handleSpeechToText(request: {
|
|
11
|
+
params: {
|
|
12
|
+
arguments: SpeechToTextRequest;
|
|
13
|
+
};
|
|
14
|
+
}, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
|
|
15
|
+
content: {
|
|
16
|
+
type: "text";
|
|
17
|
+
text: string;
|
|
18
|
+
}[];
|
|
19
|
+
_meta: Record<string, unknown>;
|
|
20
|
+
}>;
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* speech_to_text — uses OpenRouter's dedicated POST /api/v1/audio/transcriptions
|
|
3
|
+
* endpoint (launched May 2026) for speech-to-text transcription. Faster and more
|
|
4
|
+
* cost-efficient than routing through chat completions for pure transcription.
|
|
5
|
+
*
|
|
6
|
+
* Supported models: OpenAI Whisper-1, GPT-4o Transcribe, GPT-4o Mini Transcribe,
|
|
7
|
+
* Mistral Voxtral Mini Transcribe.
|
|
8
|
+
*/
|
|
9
|
+
import { promises as fs } from 'fs';
|
|
10
|
+
import path from 'node:path';
|
|
11
|
+
import { resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
12
|
+
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
13
|
+
import { SERVER_VERSION } from '../version.js';
|
|
14
|
+
import { logger } from '../logger.js';
|
|
15
|
+
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
16
|
+
import { buildCacheHeaders } from './cache.js';
|
|
17
|
+
const DEFAULT_MODEL = 'openai/whisper-1';
|
|
18
|
+
const VALID_RESPONSE_FORMATS = new Set(['json', 'text', 'srt', 'verbose_json', 'vtt']);
|
|
19
|
+
/** Infer audio format from file extension. */
|
|
20
|
+
function audioFormatFromExt(ext) {
|
|
21
|
+
const normalized = ext.toLowerCase().replace('.', '');
|
|
22
|
+
switch (normalized) {
|
|
23
|
+
case 'mp3': return 'mp3';
|
|
24
|
+
case 'mp4':
|
|
25
|
+
case 'm4a': return 'mp4';
|
|
26
|
+
case 'wav': return 'wav';
|
|
27
|
+
case 'flac': return 'flac';
|
|
28
|
+
case 'ogg':
|
|
29
|
+
case 'oga': return 'ogg';
|
|
30
|
+
case 'webm': return 'webm';
|
|
31
|
+
case 'opus': return 'opus';
|
|
32
|
+
default: return 'mp3';
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Resolve audio input to base64 + format, supporting:
|
|
37
|
+
* - data: URLs (pass through)
|
|
38
|
+
* - http(s) URLs (fetch)
|
|
39
|
+
* - local file paths (sandboxed read)
|
|
40
|
+
*/
|
|
41
|
+
async function resolveAudioInput(audioPath) {
|
|
42
|
+
const trimmed = audioPath.trim();
|
|
43
|
+
if (!trimmed)
|
|
44
|
+
throw new Error('audio_path is empty');
|
|
45
|
+
// Data URL
|
|
46
|
+
if (trimmed.startsWith('data:')) {
|
|
47
|
+
const match = trimmed.match(/^data:audio\/([^;,]+)(?:;[^,]*)*;base64,(.+)$/);
|
|
48
|
+
if (!match)
|
|
49
|
+
throw new Error('Invalid audio data URL format');
|
|
50
|
+
return { data: match[2], format: match[1] };
|
|
51
|
+
}
|
|
52
|
+
// HTTP URL
|
|
53
|
+
if (/^https?:\/\//i.test(trimmed)) {
|
|
54
|
+
const { fetchHttpResource } = await import('./fetch-utils.js');
|
|
55
|
+
const { buffer, contentType } = await fetchHttpResource(trimmed, {
|
|
56
|
+
timeoutMs: 60_000,
|
|
57
|
+
maxBytes: 100 * 1024 * 1024,
|
|
58
|
+
maxRedirects: 8,
|
|
59
|
+
});
|
|
60
|
+
const format = contentType?.match(/audio\/(\w+)/)?.[1] || 'mp3';
|
|
61
|
+
return { data: buffer.toString('base64'), format };
|
|
62
|
+
}
|
|
63
|
+
// Local file
|
|
64
|
+
const abs = await resolveSafeInputPath(trimmed);
|
|
65
|
+
const buf = await fs.readFile(abs);
|
|
66
|
+
const ext = path.extname(abs);
|
|
67
|
+
const format = audioFormatFromExt(ext);
|
|
68
|
+
return { data: buf.toString('base64'), format };
|
|
69
|
+
}
|
|
70
|
+
export async function handleSpeechToText(request, apiClient) {
|
|
71
|
+
const args = request.params.arguments ?? {};
|
|
72
|
+
const { audio_path, model, language, response_format, temperature, cache, cache_ttl, cache_clear, } = args;
|
|
73
|
+
if (!audio_path?.trim()) {
|
|
74
|
+
return toolError(ErrorCode.INVALID_INPUT, 'audio_path is required.');
|
|
75
|
+
}
|
|
76
|
+
if (response_format && !VALID_RESPONSE_FORMATS.has(response_format)) {
|
|
77
|
+
return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_RESPONSE_FORMATS].join(', ')}.`);
|
|
78
|
+
}
|
|
79
|
+
logger.audit('speech_to_text.start', {
|
|
80
|
+
model: model || DEFAULT_MODEL,
|
|
81
|
+
audio_path: audio_path.startsWith('data:') ? 'data_url' : audio_path.slice(0, 80),
|
|
82
|
+
language,
|
|
83
|
+
response_format,
|
|
84
|
+
});
|
|
85
|
+
// Resolve audio input
|
|
86
|
+
let audioInput;
|
|
87
|
+
try {
|
|
88
|
+
audioInput = await resolveAudioInput(audio_path);
|
|
89
|
+
}
|
|
90
|
+
catch (err) {
|
|
91
|
+
if (err instanceof UnsafeOutputPathError)
|
|
92
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
93
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
94
|
+
if (msg.includes('Blocked host'))
|
|
95
|
+
return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
|
|
96
|
+
return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
|
|
97
|
+
}
|
|
98
|
+
// Build request body
|
|
99
|
+
const body = {
|
|
100
|
+
model: model || DEFAULT_MODEL,
|
|
101
|
+
input_audio: {
|
|
102
|
+
data: audioInput.data,
|
|
103
|
+
format: audioInput.format,
|
|
104
|
+
},
|
|
105
|
+
};
|
|
106
|
+
if (language)
|
|
107
|
+
body.language = language;
|
|
108
|
+
if (response_format)
|
|
109
|
+
body.response_format = response_format;
|
|
110
|
+
if (typeof temperature === 'number')
|
|
111
|
+
body.temperature = temperature;
|
|
112
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
113
|
+
let response;
|
|
114
|
+
try {
|
|
115
|
+
response = await apiClient.transcribeAudio(body, headers);
|
|
116
|
+
}
|
|
117
|
+
catch (err) {
|
|
118
|
+
return classifyUpstreamError(err, 'speech_to_text');
|
|
119
|
+
}
|
|
120
|
+
const text = response.text;
|
|
121
|
+
if (!text) {
|
|
122
|
+
return toolError(ErrorCode.INTERNAL, 'Transcription returned no text.', {
|
|
123
|
+
response_keys: Object.keys(response),
|
|
124
|
+
});
|
|
125
|
+
}
|
|
126
|
+
const baseMeta = {
|
|
127
|
+
server_version: SERVER_VERSION,
|
|
128
|
+
model: model || DEFAULT_MODEL,
|
|
129
|
+
};
|
|
130
|
+
if (response.language)
|
|
131
|
+
baseMeta.language = response.language;
|
|
132
|
+
if (response.duration)
|
|
133
|
+
baseMeta.duration_seconds = response.duration;
|
|
134
|
+
if (response.usage)
|
|
135
|
+
baseMeta.usage = response.usage;
|
|
136
|
+
return {
|
|
137
|
+
content: [{ type: 'text', text }],
|
|
138
|
+
_meta: baseMeta,
|
|
139
|
+
};
|
|
140
|
+
}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import type { OpenRouterAPIClient } from '../openrouter-api.js';
|
|
2
|
+
import { type CacheOptions } from './cache.js';
|
|
3
|
+
export interface TextToSpeechRequest extends CacheOptions {
|
|
4
|
+
input: string;
|
|
5
|
+
model?: string;
|
|
6
|
+
voice?: string;
|
|
7
|
+
response_format?: string;
|
|
8
|
+
speed?: number;
|
|
9
|
+
instructions?: string;
|
|
10
|
+
save_path?: string;
|
|
11
|
+
}
|
|
12
|
+
export declare function handleTextToSpeech(request: {
|
|
13
|
+
params: {
|
|
14
|
+
arguments: TextToSpeechRequest;
|
|
15
|
+
};
|
|
16
|
+
}, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
|
|
17
|
+
content: ({
|
|
18
|
+
type: "text";
|
|
19
|
+
text: string;
|
|
20
|
+
mimeType?: undefined;
|
|
21
|
+
data?: undefined;
|
|
22
|
+
} | {
|
|
23
|
+
type: "audio";
|
|
24
|
+
mimeType: string;
|
|
25
|
+
data: string;
|
|
26
|
+
text?: undefined;
|
|
27
|
+
})[];
|
|
28
|
+
_meta: Record<string, unknown>;
|
|
29
|
+
}>;
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* text_to_speech — uses OpenRouter's dedicated POST /api/v1/audio/speech
|
|
3
|
+
* endpoint (launched May 2026) for text-to-speech. Faster and more cost-efficient
|
|
4
|
+
* than routing through chat completions with audio modality.
|
|
5
|
+
*
|
|
6
|
+
* Supported providers: OpenAI (GPT-4o Mini TTS), Google (Gemini Flash TTS),
|
|
7
|
+
* Mistral (Voxtral Mini TTS).
|
|
8
|
+
*/
|
|
9
|
+
import { promises as fs } from 'fs';
|
|
10
|
+
import { extname } from 'path';
|
|
11
|
+
import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
12
|
+
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
13
|
+
import { SERVER_VERSION } from '../version.js';
|
|
14
|
+
import { logger } from '../logger.js';
|
|
15
|
+
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
16
|
+
import { buildCacheHeaders } from './cache.js';
|
|
17
|
+
const DEFAULT_MODEL = 'openai/gpt-4o-mini-tts-2025-12-15';
|
|
18
|
+
const DEFAULT_VOICE = 'alloy';
|
|
19
|
+
const VALID_FORMATS = new Set(['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm']);
|
|
20
|
+
export async function handleTextToSpeech(request, apiClient) {
|
|
21
|
+
const args = request.params.arguments ?? {};
|
|
22
|
+
const { input, model, voice, response_format, speed, instructions, save_path, cache, cache_ttl, cache_clear, } = args;
|
|
23
|
+
if (!input?.trim()) {
|
|
24
|
+
return toolError(ErrorCode.INVALID_INPUT, 'input text is required.');
|
|
25
|
+
}
|
|
26
|
+
if (response_format && !VALID_FORMATS.has(response_format)) {
|
|
27
|
+
return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_FORMATS].join(', ')}.`);
|
|
28
|
+
}
|
|
29
|
+
logger.audit('text_to_speech.start', {
|
|
30
|
+
model: model || DEFAULT_MODEL,
|
|
31
|
+
voice: voice || DEFAULT_VOICE,
|
|
32
|
+
response_format: response_format || 'mp3',
|
|
33
|
+
input_preview: input.slice(0, 80),
|
|
34
|
+
save_path: save_path ? 'provided' : 'none',
|
|
35
|
+
});
|
|
36
|
+
// Resolve save path early
|
|
37
|
+
let safeSavePath = null;
|
|
38
|
+
if (save_path) {
|
|
39
|
+
try {
|
|
40
|
+
safeSavePath = await resolveSafeOutputPath(save_path);
|
|
41
|
+
}
|
|
42
|
+
catch (err) {
|
|
43
|
+
if (err instanceof UnsafeOutputPathError)
|
|
44
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
45
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
// Build request body
|
|
49
|
+
const body = {
|
|
50
|
+
model: model || DEFAULT_MODEL,
|
|
51
|
+
input,
|
|
52
|
+
voice: voice || DEFAULT_VOICE,
|
|
53
|
+
};
|
|
54
|
+
if (response_format)
|
|
55
|
+
body.response_format = response_format;
|
|
56
|
+
if (typeof speed === 'number' && speed > 0)
|
|
57
|
+
body.speed = speed;
|
|
58
|
+
if (instructions)
|
|
59
|
+
body.instructions = instructions;
|
|
60
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
61
|
+
let result;
|
|
62
|
+
try {
|
|
63
|
+
result = await apiClient.generateSpeech(body, headers);
|
|
64
|
+
}
|
|
65
|
+
catch (err) {
|
|
66
|
+
return classifyUpstreamError(err, 'text_to_speech');
|
|
67
|
+
}
|
|
68
|
+
const { buffer, contentType } = result;
|
|
69
|
+
const mimeType = contentType.split(';')[0]?.trim() || 'audio/mpeg';
|
|
70
|
+
// Determine file extension from format
|
|
71
|
+
const ext = response_format || 'mp3';
|
|
72
|
+
const baseMeta = {
|
|
73
|
+
server_version: SERVER_VERSION,
|
|
74
|
+
model: model || DEFAULT_MODEL,
|
|
75
|
+
mime: mimeType,
|
|
76
|
+
size_bytes: buffer.length,
|
|
77
|
+
voice: voice || DEFAULT_VOICE,
|
|
78
|
+
};
|
|
79
|
+
if (safeSavePath) {
|
|
80
|
+
// Ensure extension matches
|
|
81
|
+
const currentExt = extname(safeSavePath).toLowerCase().slice(1);
|
|
82
|
+
const actualPath = currentExt === ext ? safeSavePath : `${safeSavePath}.${ext}`;
|
|
83
|
+
try {
|
|
84
|
+
await fs.writeFile(actualPath, buffer);
|
|
85
|
+
}
|
|
86
|
+
catch (err) {
|
|
87
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
|
|
88
|
+
}
|
|
89
|
+
baseMeta.save_path = actualPath;
|
|
90
|
+
return {
|
|
91
|
+
content: [
|
|
92
|
+
{ type: 'text', text: `Speech saved to: ${actualPath}` },
|
|
93
|
+
{ type: 'audio', mimeType, data: buffer.toString('base64') },
|
|
94
|
+
],
|
|
95
|
+
_meta: baseMeta,
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
return {
|
|
99
|
+
content: [
|
|
100
|
+
{ type: 'text', text: `Speech generated (${buffer.length} bytes, ${mimeType}).` },
|
|
101
|
+
{ type: 'audio', mimeType, data: buffer.toString('base64') },
|
|
102
|
+
],
|
|
103
|
+
_meta: baseMeta,
|
|
104
|
+
};
|
|
105
|
+
}
|