@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +367 -283
- package/dist/index.js +1 -1
- package/dist/model-cache.d.ts +22 -12
- package/dist/model-cache.js +58 -21
- package/dist/openrouter-api.d.ts +45 -0
- package/dist/openrouter-api.js +50 -0
- package/dist/tool-descriptions.d.ts +19 -0
- package/dist/tool-descriptions.js +573 -0
- package/dist/tool-handlers/analyze-audio.js +5 -1
- package/dist/tool-handlers/analyze-image.js +6 -5
- package/dist/tool-handlers/analyze-video.js +6 -5
- package/dist/tool-handlers/async-chat.d.ts +51 -0
- package/dist/tool-handlers/async-chat.js +216 -0
- package/dist/tool-handlers/audio-utils.js +4 -2
- package/dist/tool-handlers/chat-completion.js +1 -1
- package/dist/tool-handlers/fetch-utils.js +16 -2
- package/dist/tool-handlers/generate-audio.js +2 -4
- package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
- package/dist/tool-handlers/generate-image-dedicated.js +176 -0
- package/dist/tool-handlers/generate-image-input.d.ts +3 -0
- package/dist/tool-handlers/generate-image-input.js +38 -0
- package/dist/tool-handlers/generate-image.d.ts +13 -51
- package/dist/tool-handlers/generate-image.js +32 -119
- package/dist/tool-handlers/generate-video.d.ts +2 -2
- package/dist/tool-handlers/generate-video.js +78 -30
- package/dist/tool-handlers/image-utils.d.ts +1 -0
- package/dist/tool-handlers/image-utils.js +26 -16
- package/dist/tool-handlers/openrouter-errors.js +6 -2
- package/dist/tool-handlers/path-safety.js +32 -5
- package/dist/tool-handlers/provider-routing.js +7 -2
- package/dist/tool-handlers/rerank.js +2 -5
- package/dist/tool-handlers/search-models.d.ts +2 -2
- package/dist/tool-handlers/search-models.js +2 -6
- package/dist/tool-handlers/speech-to-text.d.ts +20 -0
- package/dist/tool-handlers/speech-to-text.js +140 -0
- package/dist/tool-handlers/structured-output.d.ts +8 -0
- package/dist/tool-handlers/structured-output.js +11 -0
- package/dist/tool-handlers/text-to-speech.d.ts +29 -0
- package/dist/tool-handlers/text-to-speech.js +105 -0
- package/dist/tool-handlers/video-utils.js +6 -9
- package/dist/tool-handlers.js +253 -125
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +27 -15
|
@@ -14,11 +14,31 @@
|
|
|
14
14
|
*/
|
|
15
15
|
import path from 'node:path';
|
|
16
16
|
import { promises as fs } from 'node:fs';
|
|
17
|
+
import os from 'node:os';
|
|
18
|
+
/**
|
|
19
|
+
* Resolve the output root directory. On Windows, `process.cwd()` can be
|
|
20
|
+
* a system directory (e.g. `C:\Windows\System32`) when spawned by MCP
|
|
21
|
+
* clients without a working directory override. We detect that case and
|
|
22
|
+
* fall back to a writable temp directory to avoid EPERM errors.
|
|
23
|
+
*/
|
|
17
24
|
function getOutputRoot() {
|
|
18
25
|
const override = process.env.OPENROUTER_OUTPUT_DIR;
|
|
19
26
|
if (override && override.length > 0)
|
|
20
27
|
return path.resolve(override);
|
|
21
|
-
|
|
28
|
+
const cwd = process.cwd();
|
|
29
|
+
// On Windows, avoid using system directories as the default output root.
|
|
30
|
+
// Common non-writable defaults when MCP clients spawn without a cwd:
|
|
31
|
+
// C:\Windows\System32, C:\Windows, C:\Program Files\...
|
|
32
|
+
if (process.platform === 'win32') {
|
|
33
|
+
const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
|
|
34
|
+
if (cwdLower.startsWith('c:/windows') ||
|
|
35
|
+
cwdLower.startsWith('c:/program files') ||
|
|
36
|
+
cwdLower.startsWith('c:/program files (x86)')) {
|
|
37
|
+
const fallback = path.join(os.homedir(), 'openrouter-mcp-output');
|
|
38
|
+
return fallback;
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
return cwd;
|
|
22
42
|
}
|
|
23
43
|
function isUnsafeMode() {
|
|
24
44
|
const v = process.env.OPENROUTER_ALLOW_UNSAFE_PATHS;
|
|
@@ -89,9 +109,7 @@ async function findExistingAncestor(dir) {
|
|
|
89
109
|
/**
|
|
90
110
|
* Root-resolution for caller-supplied INPUT paths. Prefers
|
|
91
111
|
* `OPENROUTER_INPUT_DIR`, then `OPENROUTER_OUTPUT_DIR`, then `process.cwd()`.
|
|
92
|
-
*
|
|
93
|
-
* shipped with; exposing it here lets `generate_video`'s frame and
|
|
94
|
-
* reference images use the same sandbox.
|
|
112
|
+
* On Windows, applies the same system-directory detection as getOutputRoot.
|
|
95
113
|
*/
|
|
96
114
|
function getInputRoot() {
|
|
97
115
|
const inputDir = process.env.OPENROUTER_INPUT_DIR;
|
|
@@ -100,7 +118,16 @@ function getInputRoot() {
|
|
|
100
118
|
const outputDir = process.env.OPENROUTER_OUTPUT_DIR;
|
|
101
119
|
if (outputDir && outputDir.length > 0)
|
|
102
120
|
return path.resolve(outputDir);
|
|
103
|
-
|
|
121
|
+
const cwd = process.cwd();
|
|
122
|
+
if (process.platform === 'win32') {
|
|
123
|
+
const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
|
|
124
|
+
if (cwdLower.startsWith('c:/windows') ||
|
|
125
|
+
cwdLower.startsWith('c:/program files') ||
|
|
126
|
+
cwdLower.startsWith('c:/program files (x86)')) {
|
|
127
|
+
return path.join(os.homedir(), 'openrouter-mcp-output');
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
return cwd;
|
|
104
131
|
}
|
|
105
132
|
/**
|
|
106
133
|
* Resolve and validate a caller-supplied INPUT path. Unlike
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
* Precedence: explicit tool arg > env var > unset. Empty arrays / empty
|
|
8
8
|
* objects are dropped so we don't send noise to the API.
|
|
9
9
|
*/
|
|
10
|
+
import { logger } from '../logger.js';
|
|
10
11
|
function parseCsv(raw) {
|
|
11
12
|
if (!raw)
|
|
12
13
|
return undefined;
|
|
@@ -51,7 +52,9 @@ function parseSort(raw) {
|
|
|
51
52
|
if (!raw)
|
|
52
53
|
return undefined;
|
|
53
54
|
const lc = raw.trim().toLowerCase();
|
|
54
|
-
return lc === 'price' || lc === 'throughput' || lc === 'latency'
|
|
55
|
+
return lc === 'price' || lc === 'throughput' || lc === 'latency'
|
|
56
|
+
? lc
|
|
57
|
+
: undefined;
|
|
55
58
|
}
|
|
56
59
|
function parseDataCollection(raw) {
|
|
57
60
|
if (!raw)
|
|
@@ -85,7 +88,9 @@ export function readProviderDefaults() {
|
|
|
85
88
|
// operator notices instead of wondering why their ordering is being
|
|
86
89
|
// ignored. All other OPENROUTER_PROVIDER_* fields follow the same
|
|
87
90
|
// "silent drop" policy for consistency.
|
|
88
|
-
|
|
91
|
+
logger.warn('OPENROUTER_PROVIDER_ORDER ignored', {
|
|
92
|
+
err: err instanceof Error ? err.message : String(err),
|
|
93
|
+
});
|
|
89
94
|
}
|
|
90
95
|
const requireParams = parseBool(env.OPENROUTER_PROVIDER_REQUIRE_PARAMETERS);
|
|
91
96
|
if (requireParams !== undefined)
|
|
@@ -3,8 +3,7 @@ import { classifyUpstreamError } from './openrouter-errors.js';
|
|
|
3
3
|
import { buildStructuredResult } from './structured-output.js';
|
|
4
4
|
const DEFAULT_MODEL = 'cohere/rerank-english-v3.0';
|
|
5
5
|
export async function handleRerankDocuments(request, apiClient) {
|
|
6
|
-
const args = request.params.arguments ??
|
|
7
|
-
{ query: '', documents: [] };
|
|
6
|
+
const args = request.params.arguments ?? { query: '', documents: [] };
|
|
8
7
|
const { query, documents, model, top_n, return_documents } = args;
|
|
9
8
|
if (!query?.trim()) {
|
|
10
9
|
return toolError(ErrorCode.INVALID_INPUT, 'query is required.');
|
|
@@ -33,9 +32,7 @@ export async function handleRerankDocuments(request, apiClient) {
|
|
|
33
32
|
const score = typeof r.score === 'number' ? r.score : r.relevance_score;
|
|
34
33
|
const out = { index: r.index, score };
|
|
35
34
|
if (return_documents) {
|
|
36
|
-
const doc = typeof r.document === 'string'
|
|
37
|
-
? r.document
|
|
38
|
-
: r.document?.text ?? documents[r.index];
|
|
35
|
+
const doc = typeof r.document === 'string' ? r.document : (r.document?.text ?? documents[r.index]);
|
|
39
36
|
out.document = doc;
|
|
40
37
|
}
|
|
41
38
|
return out;
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { ModelCache
|
|
1
|
+
import { ModelCache } from '../model-cache.js';
|
|
2
2
|
import { OpenRouterAPIClient } from '../openrouter-api.js';
|
|
3
3
|
export interface SearchModelsArgs {
|
|
4
4
|
query?: string;
|
|
@@ -22,7 +22,7 @@ export declare function handleSearchModels(request: {
|
|
|
22
22
|
arguments: SearchModelsArgs;
|
|
23
23
|
};
|
|
24
24
|
}, apiClient: OpenRouterAPIClient, modelCache: ModelCache): Promise<import("../errors.js").ToolErrorResult | import("./structured-output.js").StructuredResult<{
|
|
25
|
-
results: OpenRouterModelRecord[];
|
|
25
|
+
results: import("../model-cache.js").OpenRouterModelRecord[];
|
|
26
26
|
offset: number;
|
|
27
27
|
limit: number;
|
|
28
28
|
total: number;
|
|
@@ -14,15 +14,11 @@ export async function handleSearchModels(request, apiClient, modelCache) {
|
|
|
14
14
|
const args = request.params.arguments ?? {};
|
|
15
15
|
const limit = Math.min(Math.max(1, args.limit ?? DEFAULT_LIMIT), MAX_LIMIT);
|
|
16
16
|
const offset = Math.max(0, args.offset ?? 0);
|
|
17
|
-
|
|
18
|
-
const all = modelCache.search({
|
|
17
|
+
const { page, total } = modelCache.searchPaginated({
|
|
19
18
|
query: args.query,
|
|
20
19
|
provider: args.provider,
|
|
21
20
|
capabilities: args.capabilities,
|
|
22
|
-
|
|
23
|
-
});
|
|
24
|
-
const total = all.length;
|
|
25
|
-
const page = all.slice(offset, offset + limit);
|
|
21
|
+
}, offset, limit);
|
|
26
22
|
const nextOffset = offset + limit;
|
|
27
23
|
const hasMore = nextOffset < total;
|
|
28
24
|
return buildStructuredResult({
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
import type { OpenRouterAPIClient } from '../openrouter-api.js';
|
|
2
|
+
import { type CacheOptions } from './cache.js';
|
|
3
|
+
export interface SpeechToTextRequest extends CacheOptions {
|
|
4
|
+
audio_path: string;
|
|
5
|
+
model?: string;
|
|
6
|
+
language?: string;
|
|
7
|
+
response_format?: string;
|
|
8
|
+
temperature?: number;
|
|
9
|
+
}
|
|
10
|
+
export declare function handleSpeechToText(request: {
|
|
11
|
+
params: {
|
|
12
|
+
arguments: SpeechToTextRequest;
|
|
13
|
+
};
|
|
14
|
+
}, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
|
|
15
|
+
content: {
|
|
16
|
+
type: "text";
|
|
17
|
+
text: string;
|
|
18
|
+
}[];
|
|
19
|
+
_meta: Record<string, unknown>;
|
|
20
|
+
}>;
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* speech_to_text — uses OpenRouter's dedicated POST /api/v1/audio/transcriptions
|
|
3
|
+
* endpoint (launched May 2026) for speech-to-text transcription. Faster and more
|
|
4
|
+
* cost-efficient than routing through chat completions for pure transcription.
|
|
5
|
+
*
|
|
6
|
+
* Supported models: OpenAI Whisper-1, GPT-4o Transcribe, GPT-4o Mini Transcribe,
|
|
7
|
+
* Mistral Voxtral Mini Transcribe.
|
|
8
|
+
*/
|
|
9
|
+
import { promises as fs } from 'fs';
|
|
10
|
+
import path from 'node:path';
|
|
11
|
+
import { resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
12
|
+
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
13
|
+
import { SERVER_VERSION } from '../version.js';
|
|
14
|
+
import { logger } from '../logger.js';
|
|
15
|
+
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
16
|
+
import { buildCacheHeaders } from './cache.js';
|
|
17
|
+
const DEFAULT_MODEL = 'openai/whisper-1';
|
|
18
|
+
const VALID_RESPONSE_FORMATS = new Set(['json', 'text', 'srt', 'verbose_json', 'vtt']);
|
|
19
|
+
/** Infer audio format from file extension. */
|
|
20
|
+
function audioFormatFromExt(ext) {
|
|
21
|
+
const normalized = ext.toLowerCase().replace('.', '');
|
|
22
|
+
switch (normalized) {
|
|
23
|
+
case 'mp3': return 'mp3';
|
|
24
|
+
case 'mp4':
|
|
25
|
+
case 'm4a': return 'mp4';
|
|
26
|
+
case 'wav': return 'wav';
|
|
27
|
+
case 'flac': return 'flac';
|
|
28
|
+
case 'ogg':
|
|
29
|
+
case 'oga': return 'ogg';
|
|
30
|
+
case 'webm': return 'webm';
|
|
31
|
+
case 'opus': return 'opus';
|
|
32
|
+
default: return 'mp3';
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Resolve audio input to base64 + format, supporting:
|
|
37
|
+
* - data: URLs (pass through)
|
|
38
|
+
* - http(s) URLs (fetch)
|
|
39
|
+
* - local file paths (sandboxed read)
|
|
40
|
+
*/
|
|
41
|
+
async function resolveAudioInput(audioPath) {
|
|
42
|
+
const trimmed = audioPath.trim();
|
|
43
|
+
if (!trimmed)
|
|
44
|
+
throw new Error('audio_path is empty');
|
|
45
|
+
// Data URL
|
|
46
|
+
if (trimmed.startsWith('data:')) {
|
|
47
|
+
const match = trimmed.match(/^data:audio\/([^;,]+)(?:;[^,]*)*;base64,(.+)$/);
|
|
48
|
+
if (!match)
|
|
49
|
+
throw new Error('Invalid audio data URL format');
|
|
50
|
+
return { data: match[2], format: match[1] };
|
|
51
|
+
}
|
|
52
|
+
// HTTP URL
|
|
53
|
+
if (/^https?:\/\//i.test(trimmed)) {
|
|
54
|
+
const { fetchHttpResource } = await import('./fetch-utils.js');
|
|
55
|
+
const { buffer, contentType } = await fetchHttpResource(trimmed, {
|
|
56
|
+
timeoutMs: 60_000,
|
|
57
|
+
maxBytes: 100 * 1024 * 1024,
|
|
58
|
+
maxRedirects: 8,
|
|
59
|
+
});
|
|
60
|
+
const format = contentType?.match(/audio\/(\w+)/)?.[1] || 'mp3';
|
|
61
|
+
return { data: buffer.toString('base64'), format };
|
|
62
|
+
}
|
|
63
|
+
// Local file
|
|
64
|
+
const abs = await resolveSafeInputPath(trimmed);
|
|
65
|
+
const buf = await fs.readFile(abs);
|
|
66
|
+
const ext = path.extname(abs);
|
|
67
|
+
const format = audioFormatFromExt(ext);
|
|
68
|
+
return { data: buf.toString('base64'), format };
|
|
69
|
+
}
|
|
70
|
+
export async function handleSpeechToText(request, apiClient) {
|
|
71
|
+
const args = request.params.arguments ?? {};
|
|
72
|
+
const { audio_path, model, language, response_format, temperature, cache, cache_ttl, cache_clear, } = args;
|
|
73
|
+
if (!audio_path?.trim()) {
|
|
74
|
+
return toolError(ErrorCode.INVALID_INPUT, 'audio_path is required.');
|
|
75
|
+
}
|
|
76
|
+
if (response_format && !VALID_RESPONSE_FORMATS.has(response_format)) {
|
|
77
|
+
return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_RESPONSE_FORMATS].join(', ')}.`);
|
|
78
|
+
}
|
|
79
|
+
logger.audit('speech_to_text.start', {
|
|
80
|
+
model: model || DEFAULT_MODEL,
|
|
81
|
+
audio_path: audio_path.startsWith('data:') ? 'data_url' : audio_path.slice(0, 80),
|
|
82
|
+
language,
|
|
83
|
+
response_format,
|
|
84
|
+
});
|
|
85
|
+
// Resolve audio input
|
|
86
|
+
let audioInput;
|
|
87
|
+
try {
|
|
88
|
+
audioInput = await resolveAudioInput(audio_path);
|
|
89
|
+
}
|
|
90
|
+
catch (err) {
|
|
91
|
+
if (err instanceof UnsafeOutputPathError)
|
|
92
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
93
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
94
|
+
if (msg.includes('Blocked host'))
|
|
95
|
+
return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
|
|
96
|
+
return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
|
|
97
|
+
}
|
|
98
|
+
// Build request body
|
|
99
|
+
const body = {
|
|
100
|
+
model: model || DEFAULT_MODEL,
|
|
101
|
+
input_audio: {
|
|
102
|
+
data: audioInput.data,
|
|
103
|
+
format: audioInput.format,
|
|
104
|
+
},
|
|
105
|
+
};
|
|
106
|
+
if (language)
|
|
107
|
+
body.language = language;
|
|
108
|
+
if (response_format)
|
|
109
|
+
body.response_format = response_format;
|
|
110
|
+
if (typeof temperature === 'number')
|
|
111
|
+
body.temperature = temperature;
|
|
112
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
113
|
+
let response;
|
|
114
|
+
try {
|
|
115
|
+
response = await apiClient.transcribeAudio(body, headers);
|
|
116
|
+
}
|
|
117
|
+
catch (err) {
|
|
118
|
+
return classifyUpstreamError(err, 'speech_to_text');
|
|
119
|
+
}
|
|
120
|
+
const text = response.text;
|
|
121
|
+
if (!text) {
|
|
122
|
+
return toolError(ErrorCode.INTERNAL, 'Transcription returned no text.', {
|
|
123
|
+
response_keys: Object.keys(response),
|
|
124
|
+
});
|
|
125
|
+
}
|
|
126
|
+
const baseMeta = {
|
|
127
|
+
server_version: SERVER_VERSION,
|
|
128
|
+
model: model || DEFAULT_MODEL,
|
|
129
|
+
};
|
|
130
|
+
if (response.language)
|
|
131
|
+
baseMeta.language = response.language;
|
|
132
|
+
if (response.duration)
|
|
133
|
+
baseMeta.duration_seconds = response.duration;
|
|
134
|
+
if (response.usage)
|
|
135
|
+
baseMeta.usage = response.usage;
|
|
136
|
+
return {
|
|
137
|
+
content: [{ type: 'text', text }],
|
|
138
|
+
_meta: baseMeta,
|
|
139
|
+
};
|
|
140
|
+
}
|
|
@@ -11,3 +11,11 @@ export interface StructuredResult<T = unknown> {
|
|
|
11
11
|
* format. `meta` is merged on top of the default `server_version` stamp.
|
|
12
12
|
*/
|
|
13
13
|
export declare function buildStructuredResult<T>(data: T, meta?: Record<string, unknown>): StructuredResult<T>;
|
|
14
|
+
/** Read typed JSON from an MCP tool result (structuredContent or legacy text). */
|
|
15
|
+
export declare function readToolPayload<T = unknown>(result: {
|
|
16
|
+
structuredContent?: T;
|
|
17
|
+
content?: Array<{
|
|
18
|
+
type: string;
|
|
19
|
+
text?: string;
|
|
20
|
+
}>;
|
|
21
|
+
}): T;
|
|
@@ -22,3 +22,14 @@ export function buildStructuredResult(data, meta = {}) {
|
|
|
22
22
|
_meta: { server_version: SERVER_VERSION, ...meta },
|
|
23
23
|
};
|
|
24
24
|
}
|
|
25
|
+
/** Read typed JSON from an MCP tool result (structuredContent or legacy text). */
|
|
26
|
+
export function readToolPayload(result) {
|
|
27
|
+
if (result.structuredContent !== undefined) {
|
|
28
|
+
return result.structuredContent;
|
|
29
|
+
}
|
|
30
|
+
const text = result.content?.[0]?.text;
|
|
31
|
+
if (text === undefined) {
|
|
32
|
+
throw new Error('tool result has no structuredContent or content text');
|
|
33
|
+
}
|
|
34
|
+
return JSON.parse(text);
|
|
35
|
+
}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import type { OpenRouterAPIClient } from '../openrouter-api.js';
|
|
2
|
+
import { type CacheOptions } from './cache.js';
|
|
3
|
+
export interface TextToSpeechRequest extends CacheOptions {
|
|
4
|
+
input: string;
|
|
5
|
+
model?: string;
|
|
6
|
+
voice?: string;
|
|
7
|
+
response_format?: string;
|
|
8
|
+
speed?: number;
|
|
9
|
+
instructions?: string;
|
|
10
|
+
save_path?: string;
|
|
11
|
+
}
|
|
12
|
+
export declare function handleTextToSpeech(request: {
|
|
13
|
+
params: {
|
|
14
|
+
arguments: TextToSpeechRequest;
|
|
15
|
+
};
|
|
16
|
+
}, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
|
|
17
|
+
content: ({
|
|
18
|
+
type: "text";
|
|
19
|
+
text: string;
|
|
20
|
+
mimeType?: undefined;
|
|
21
|
+
data?: undefined;
|
|
22
|
+
} | {
|
|
23
|
+
type: "audio";
|
|
24
|
+
mimeType: string;
|
|
25
|
+
data: string;
|
|
26
|
+
text?: undefined;
|
|
27
|
+
})[];
|
|
28
|
+
_meta: Record<string, unknown>;
|
|
29
|
+
}>;
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* text_to_speech — uses OpenRouter's dedicated POST /api/v1/audio/speech
|
|
3
|
+
* endpoint (launched May 2026) for text-to-speech. Faster and more cost-efficient
|
|
4
|
+
* than routing through chat completions with audio modality.
|
|
5
|
+
*
|
|
6
|
+
* Supported providers: OpenAI (GPT-4o Mini TTS), Google (Gemini Flash TTS),
|
|
7
|
+
* Mistral (Voxtral Mini TTS).
|
|
8
|
+
*/
|
|
9
|
+
import { promises as fs } from 'fs';
|
|
10
|
+
import { extname } from 'path';
|
|
11
|
+
import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
|
|
12
|
+
import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
|
|
13
|
+
import { SERVER_VERSION } from '../version.js';
|
|
14
|
+
import { logger } from '../logger.js';
|
|
15
|
+
import { classifyUpstreamError } from './openrouter-errors.js';
|
|
16
|
+
import { buildCacheHeaders } from './cache.js';
|
|
17
|
+
const DEFAULT_MODEL = 'openai/gpt-4o-mini-tts-2025-12-15';
|
|
18
|
+
const DEFAULT_VOICE = 'alloy';
|
|
19
|
+
const VALID_FORMATS = new Set(['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm']);
|
|
20
|
+
export async function handleTextToSpeech(request, apiClient) {
|
|
21
|
+
const args = request.params.arguments ?? {};
|
|
22
|
+
const { input, model, voice, response_format, speed, instructions, save_path, cache, cache_ttl, cache_clear, } = args;
|
|
23
|
+
if (!input?.trim()) {
|
|
24
|
+
return toolError(ErrorCode.INVALID_INPUT, 'input text is required.');
|
|
25
|
+
}
|
|
26
|
+
if (response_format && !VALID_FORMATS.has(response_format)) {
|
|
27
|
+
return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_FORMATS].join(', ')}.`);
|
|
28
|
+
}
|
|
29
|
+
logger.audit('text_to_speech.start', {
|
|
30
|
+
model: model || DEFAULT_MODEL,
|
|
31
|
+
voice: voice || DEFAULT_VOICE,
|
|
32
|
+
response_format: response_format || 'mp3',
|
|
33
|
+
input_preview: input.slice(0, 80),
|
|
34
|
+
save_path: save_path ? 'provided' : 'none',
|
|
35
|
+
});
|
|
36
|
+
// Resolve save path early
|
|
37
|
+
let safeSavePath = null;
|
|
38
|
+
if (save_path) {
|
|
39
|
+
try {
|
|
40
|
+
safeSavePath = await resolveSafeOutputPath(save_path);
|
|
41
|
+
}
|
|
42
|
+
catch (err) {
|
|
43
|
+
if (err instanceof UnsafeOutputPathError)
|
|
44
|
+
return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
|
|
45
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
// Build request body
|
|
49
|
+
const body = {
|
|
50
|
+
model: model || DEFAULT_MODEL,
|
|
51
|
+
input,
|
|
52
|
+
voice: voice || DEFAULT_VOICE,
|
|
53
|
+
};
|
|
54
|
+
if (response_format)
|
|
55
|
+
body.response_format = response_format;
|
|
56
|
+
if (typeof speed === 'number' && speed > 0)
|
|
57
|
+
body.speed = speed;
|
|
58
|
+
if (instructions)
|
|
59
|
+
body.instructions = instructions;
|
|
60
|
+
const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
|
|
61
|
+
let result;
|
|
62
|
+
try {
|
|
63
|
+
result = await apiClient.generateSpeech(body, headers);
|
|
64
|
+
}
|
|
65
|
+
catch (err) {
|
|
66
|
+
return classifyUpstreamError(err, 'text_to_speech');
|
|
67
|
+
}
|
|
68
|
+
const { buffer, contentType } = result;
|
|
69
|
+
const mimeType = contentType.split(';')[0]?.trim() || 'audio/mpeg';
|
|
70
|
+
// Determine file extension from format
|
|
71
|
+
const ext = response_format || 'mp3';
|
|
72
|
+
const baseMeta = {
|
|
73
|
+
server_version: SERVER_VERSION,
|
|
74
|
+
model: model || DEFAULT_MODEL,
|
|
75
|
+
mime: mimeType,
|
|
76
|
+
size_bytes: buffer.length,
|
|
77
|
+
voice: voice || DEFAULT_VOICE,
|
|
78
|
+
};
|
|
79
|
+
if (safeSavePath) {
|
|
80
|
+
// Ensure extension matches
|
|
81
|
+
const currentExt = extname(safeSavePath).toLowerCase().slice(1);
|
|
82
|
+
const actualPath = currentExt === ext ? safeSavePath : `${safeSavePath}.${ext}`;
|
|
83
|
+
try {
|
|
84
|
+
await fs.writeFile(actualPath, buffer);
|
|
85
|
+
}
|
|
86
|
+
catch (err) {
|
|
87
|
+
return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
|
|
88
|
+
}
|
|
89
|
+
baseMeta.save_path = actualPath;
|
|
90
|
+
return {
|
|
91
|
+
content: [
|
|
92
|
+
{ type: 'text', text: `Speech saved to: ${actualPath}` },
|
|
93
|
+
{ type: 'audio', mimeType, data: buffer.toString('base64') },
|
|
94
|
+
],
|
|
95
|
+
_meta: baseMeta,
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
return {
|
|
99
|
+
content: [
|
|
100
|
+
{ type: 'text', text: `Speech generated (${buffer.length} bytes, ${mimeType}).` },
|
|
101
|
+
{ type: 'audio', mimeType, data: buffer.toString('base64') },
|
|
102
|
+
],
|
|
103
|
+
_meta: baseMeta,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
import path from 'node:path';
|
|
10
10
|
import { promises as fs } from 'node:fs';
|
|
11
11
|
import { readEnvInt, fetchHttpResource, parseBase64DataUrl } from './fetch-utils.js';
|
|
12
|
+
import { resolveSafeInputPath } from './path-safety.js';
|
|
12
13
|
export { isBlockedIPv4, assertUrlSafeForFetch } from './fetch-utils.js';
|
|
13
14
|
const DEFAULT_FETCH_TIMEOUT_MS = 60_000;
|
|
14
15
|
const DEFAULT_MAX_DOWNLOAD_BYTES = 100 * 1024 * 1024; // 100 MB
|
|
@@ -93,10 +94,7 @@ export function detectVideoFormat(buffer) {
|
|
|
93
94
|
}
|
|
94
95
|
if (buffer.length >= 4) {
|
|
95
96
|
// EBML header — WebM & Matroska.
|
|
96
|
-
if (buffer[0] === 0x1a &&
|
|
97
|
-
buffer[1] === 0x45 &&
|
|
98
|
-
buffer[2] === 0xdf &&
|
|
99
|
-
buffer[3] === 0xa3) {
|
|
97
|
+
if (buffer[0] === 0x1a && buffer[1] === 0x45 && buffer[2] === 0xdf && buffer[3] === 0xa3) {
|
|
100
98
|
return 'webm';
|
|
101
99
|
}
|
|
102
100
|
// MPEG-PS / MPEG-TS start codes.
|
|
@@ -146,9 +144,7 @@ export async function prepareVideoData(source) {
|
|
|
146
144
|
maxRedirects: getMaxRedirects(),
|
|
147
145
|
});
|
|
148
146
|
const urlPath = new URL(source).pathname;
|
|
149
|
-
const format = detectVideoFormat(buffer) ??
|
|
150
|
-
getVideoFormat(urlPath) ??
|
|
151
|
-
formatFromContentType(contentType);
|
|
147
|
+
const format = detectVideoFormat(buffer) ?? getVideoFormat(urlPath) ?? formatFromContentType(contentType);
|
|
152
148
|
if (!format) {
|
|
153
149
|
throw new Error(`Could not determine video format from ${source}. Supported: ${SUPPORTED_VIDEO_FORMATS.join(', ')}`);
|
|
154
150
|
}
|
|
@@ -160,8 +156,9 @@ export async function prepareVideoData(source) {
|
|
|
160
156
|
};
|
|
161
157
|
}
|
|
162
158
|
// --- local file ---
|
|
163
|
-
const
|
|
164
|
-
const
|
|
159
|
+
const safe = await resolveSafeInputPath(source);
|
|
160
|
+
const buffer = await fs.readFile(safe);
|
|
161
|
+
const format = detectVideoFormat(buffer) ?? getVideoFormat(safe);
|
|
165
162
|
if (!format) {
|
|
166
163
|
throw new Error(`Unsupported video format for file: ${source}. Supported: ${SUPPORTED_VIDEO_FORMATS.join(', ')}`);
|
|
167
164
|
}
|