@stabgan/openrouter-mcp-multimodal 4.5.1 → 4.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +367 -283
  2. package/dist/index.js +1 -1
  3. package/dist/model-cache.d.ts +22 -12
  4. package/dist/model-cache.js +58 -21
  5. package/dist/openrouter-api.d.ts +45 -0
  6. package/dist/openrouter-api.js +50 -0
  7. package/dist/tool-descriptions.d.ts +19 -0
  8. package/dist/tool-descriptions.js +573 -0
  9. package/dist/tool-handlers/analyze-audio.js +5 -1
  10. package/dist/tool-handlers/analyze-image.js +6 -5
  11. package/dist/tool-handlers/analyze-video.js +6 -5
  12. package/dist/tool-handlers/async-chat.d.ts +51 -0
  13. package/dist/tool-handlers/async-chat.js +216 -0
  14. package/dist/tool-handlers/audio-utils.js +4 -2
  15. package/dist/tool-handlers/chat-completion.js +1 -1
  16. package/dist/tool-handlers/fetch-utils.js +16 -2
  17. package/dist/tool-handlers/generate-audio.js +2 -4
  18. package/dist/tool-handlers/generate-image-dedicated.d.ts +32 -0
  19. package/dist/tool-handlers/generate-image-dedicated.js +176 -0
  20. package/dist/tool-handlers/generate-image-input.d.ts +3 -0
  21. package/dist/tool-handlers/generate-image-input.js +38 -0
  22. package/dist/tool-handlers/generate-image.d.ts +13 -51
  23. package/dist/tool-handlers/generate-image.js +32 -119
  24. package/dist/tool-handlers/generate-video.d.ts +2 -2
  25. package/dist/tool-handlers/generate-video.js +78 -30
  26. package/dist/tool-handlers/image-utils.d.ts +1 -0
  27. package/dist/tool-handlers/image-utils.js +26 -16
  28. package/dist/tool-handlers/openrouter-errors.js +6 -2
  29. package/dist/tool-handlers/path-safety.js +32 -5
  30. package/dist/tool-handlers/provider-routing.js +7 -2
  31. package/dist/tool-handlers/rerank.js +2 -5
  32. package/dist/tool-handlers/search-models.d.ts +2 -2
  33. package/dist/tool-handlers/search-models.js +2 -6
  34. package/dist/tool-handlers/speech-to-text.d.ts +20 -0
  35. package/dist/tool-handlers/speech-to-text.js +140 -0
  36. package/dist/tool-handlers/structured-output.d.ts +8 -0
  37. package/dist/tool-handlers/structured-output.js +11 -0
  38. package/dist/tool-handlers/text-to-speech.d.ts +29 -0
  39. package/dist/tool-handlers/text-to-speech.js +105 -0
  40. package/dist/tool-handlers/video-utils.js +6 -9
  41. package/dist/tool-handlers.js +253 -125
  42. package/dist/version.d.ts +1 -1
  43. package/dist/version.js +1 -1
  44. package/package.json +27 -15
@@ -14,11 +14,31 @@
14
14
  */
15
15
  import path from 'node:path';
16
16
  import { promises as fs } from 'node:fs';
17
+ import os from 'node:os';
18
+ /**
19
+ * Resolve the output root directory. On Windows, `process.cwd()` can be
20
+ * a system directory (e.g. `C:\Windows\System32`) when spawned by MCP
21
+ * clients without a working directory override. We detect that case and
22
+ * fall back to a writable temp directory to avoid EPERM errors.
23
+ */
17
24
  function getOutputRoot() {
18
25
  const override = process.env.OPENROUTER_OUTPUT_DIR;
19
26
  if (override && override.length > 0)
20
27
  return path.resolve(override);
21
- return process.cwd();
28
+ const cwd = process.cwd();
29
+ // On Windows, avoid using system directories as the default output root.
30
+ // Common non-writable defaults when MCP clients spawn without a cwd:
31
+ // C:\Windows\System32, C:\Windows, C:\Program Files\...
32
+ if (process.platform === 'win32') {
33
+ const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
34
+ if (cwdLower.startsWith('c:/windows') ||
35
+ cwdLower.startsWith('c:/program files') ||
36
+ cwdLower.startsWith('c:/program files (x86)')) {
37
+ const fallback = path.join(os.homedir(), 'openrouter-mcp-output');
38
+ return fallback;
39
+ }
40
+ }
41
+ return cwd;
22
42
  }
23
43
  function isUnsafeMode() {
24
44
  const v = process.env.OPENROUTER_ALLOW_UNSAFE_PATHS;
@@ -89,9 +109,7 @@ async function findExistingAncestor(dir) {
89
109
  /**
90
110
  * Root-resolution for caller-supplied INPUT paths. Prefers
91
111
  * `OPENROUTER_INPUT_DIR`, then `OPENROUTER_OUTPUT_DIR`, then `process.cwd()`.
92
- * This mirrors the semantics `generate_image`'s `input_images` originally
93
- * shipped with; exposing it here lets `generate_video`'s frame and
94
- * reference images use the same sandbox.
112
+ * On Windows, applies the same system-directory detection as getOutputRoot.
95
113
  */
96
114
  function getInputRoot() {
97
115
  const inputDir = process.env.OPENROUTER_INPUT_DIR;
@@ -100,7 +118,16 @@ function getInputRoot() {
100
118
  const outputDir = process.env.OPENROUTER_OUTPUT_DIR;
101
119
  if (outputDir && outputDir.length > 0)
102
120
  return path.resolve(outputDir);
103
- return process.cwd();
121
+ const cwd = process.cwd();
122
+ if (process.platform === 'win32') {
123
+ const cwdLower = cwd.toLowerCase().replace(/\\/g, '/');
124
+ if (cwdLower.startsWith('c:/windows') ||
125
+ cwdLower.startsWith('c:/program files') ||
126
+ cwdLower.startsWith('c:/program files (x86)')) {
127
+ return path.join(os.homedir(), 'openrouter-mcp-output');
128
+ }
129
+ }
130
+ return cwd;
104
131
  }
105
132
  /**
106
133
  * Resolve and validate a caller-supplied INPUT path. Unlike
@@ -7,6 +7,7 @@
7
7
  * Precedence: explicit tool arg > env var > unset. Empty arrays / empty
8
8
  * objects are dropped so we don't send noise to the API.
9
9
  */
10
+ import { logger } from '../logger.js';
10
11
  function parseCsv(raw) {
11
12
  if (!raw)
12
13
  return undefined;
@@ -51,7 +52,9 @@ function parseSort(raw) {
51
52
  if (!raw)
52
53
  return undefined;
53
54
  const lc = raw.trim().toLowerCase();
54
- return lc === 'price' || lc === 'throughput' || lc === 'latency' ? lc : undefined;
55
+ return lc === 'price' || lc === 'throughput' || lc === 'latency'
56
+ ? lc
57
+ : undefined;
55
58
  }
56
59
  function parseDataCollection(raw) {
57
60
  if (!raw)
@@ -85,7 +88,9 @@ export function readProviderDefaults() {
85
88
  // operator notices instead of wondering why their ordering is being
86
89
  // ignored. All other OPENROUTER_PROVIDER_* fields follow the same
87
90
  // "silent drop" policy for consistency.
88
- console.error(`[openrouter-mcp] OPENROUTER_PROVIDER_ORDER ignored: ${err instanceof Error ? err.message : String(err)}`);
91
+ logger.warn('OPENROUTER_PROVIDER_ORDER ignored', {
92
+ err: err instanceof Error ? err.message : String(err),
93
+ });
89
94
  }
90
95
  const requireParams = parseBool(env.OPENROUTER_PROVIDER_REQUIRE_PARAMETERS);
91
96
  if (requireParams !== undefined)
@@ -3,8 +3,7 @@ import { classifyUpstreamError } from './openrouter-errors.js';
3
3
  import { buildStructuredResult } from './structured-output.js';
4
4
  const DEFAULT_MODEL = 'cohere/rerank-english-v3.0';
5
5
  export async function handleRerankDocuments(request, apiClient) {
6
- const args = request.params.arguments ??
7
- { query: '', documents: [] };
6
+ const args = request.params.arguments ?? { query: '', documents: [] };
8
7
  const { query, documents, model, top_n, return_documents } = args;
9
8
  if (!query?.trim()) {
10
9
  return toolError(ErrorCode.INVALID_INPUT, 'query is required.');
@@ -33,9 +32,7 @@ export async function handleRerankDocuments(request, apiClient) {
33
32
  const score = typeof r.score === 'number' ? r.score : r.relevance_score;
34
33
  const out = { index: r.index, score };
35
34
  if (return_documents) {
36
- const doc = typeof r.document === 'string'
37
- ? r.document
38
- : r.document?.text ?? documents[r.index];
35
+ const doc = typeof r.document === 'string' ? r.document : (r.document?.text ?? documents[r.index]);
39
36
  out.document = doc;
40
37
  }
41
38
  return out;
@@ -1,4 +1,4 @@
1
- import { ModelCache, type OpenRouterModelRecord } from '../model-cache.js';
1
+ import { ModelCache } from '../model-cache.js';
2
2
  import { OpenRouterAPIClient } from '../openrouter-api.js';
3
3
  export interface SearchModelsArgs {
4
4
  query?: string;
@@ -22,7 +22,7 @@ export declare function handleSearchModels(request: {
22
22
  arguments: SearchModelsArgs;
23
23
  };
24
24
  }, apiClient: OpenRouterAPIClient, modelCache: ModelCache): Promise<import("../errors.js").ToolErrorResult | import("./structured-output.js").StructuredResult<{
25
- results: OpenRouterModelRecord[];
25
+ results: import("../model-cache.js").OpenRouterModelRecord[];
26
26
  offset: number;
27
27
  limit: number;
28
28
  total: number;
@@ -14,15 +14,11 @@ export async function handleSearchModels(request, apiClient, modelCache) {
14
14
  const args = request.params.arguments ?? {};
15
15
  const limit = Math.min(Math.max(1, args.limit ?? DEFAULT_LIMIT), MAX_LIMIT);
16
16
  const offset = Math.max(0, args.offset ?? 0);
17
- // Get the full filtered set, then slice for pagination.
18
- const all = modelCache.search({
17
+ const { page, total } = modelCache.searchPaginated({
19
18
  query: args.query,
20
19
  provider: args.provider,
21
20
  capabilities: args.capabilities,
22
- all: true,
23
- });
24
- const total = all.length;
25
- const page = all.slice(offset, offset + limit);
21
+ }, offset, limit);
26
22
  const nextOffset = offset + limit;
27
23
  const hasMore = nextOffset < total;
28
24
  return buildStructuredResult({
@@ -0,0 +1,20 @@
1
+ import type { OpenRouterAPIClient } from '../openrouter-api.js';
2
+ import { type CacheOptions } from './cache.js';
3
+ export interface SpeechToTextRequest extends CacheOptions {
4
+ audio_path: string;
5
+ model?: string;
6
+ language?: string;
7
+ response_format?: string;
8
+ temperature?: number;
9
+ }
10
+ export declare function handleSpeechToText(request: {
11
+ params: {
12
+ arguments: SpeechToTextRequest;
13
+ };
14
+ }, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
15
+ content: {
16
+ type: "text";
17
+ text: string;
18
+ }[];
19
+ _meta: Record<string, unknown>;
20
+ }>;
@@ -0,0 +1,140 @@
1
+ /**
2
+ * speech_to_text — uses OpenRouter's dedicated POST /api/v1/audio/transcriptions
3
+ * endpoint (launched May 2026) for speech-to-text transcription. Faster and more
4
+ * cost-efficient than routing through chat completions for pure transcription.
5
+ *
6
+ * Supported models: OpenAI Whisper-1, GPT-4o Transcribe, GPT-4o Mini Transcribe,
7
+ * Mistral Voxtral Mini Transcribe.
8
+ */
9
+ import { promises as fs } from 'fs';
10
+ import path from 'node:path';
11
+ import { resolveSafeInputPath, UnsafeOutputPathError } from './path-safety.js';
12
+ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
13
+ import { SERVER_VERSION } from '../version.js';
14
+ import { logger } from '../logger.js';
15
+ import { classifyUpstreamError } from './openrouter-errors.js';
16
+ import { buildCacheHeaders } from './cache.js';
17
+ const DEFAULT_MODEL = 'openai/whisper-1';
18
+ const VALID_RESPONSE_FORMATS = new Set(['json', 'text', 'srt', 'verbose_json', 'vtt']);
19
+ /** Infer audio format from file extension. */
20
+ function audioFormatFromExt(ext) {
21
+ const normalized = ext.toLowerCase().replace('.', '');
22
+ switch (normalized) {
23
+ case 'mp3': return 'mp3';
24
+ case 'mp4':
25
+ case 'm4a': return 'mp4';
26
+ case 'wav': return 'wav';
27
+ case 'flac': return 'flac';
28
+ case 'ogg':
29
+ case 'oga': return 'ogg';
30
+ case 'webm': return 'webm';
31
+ case 'opus': return 'opus';
32
+ default: return 'mp3';
33
+ }
34
+ }
35
+ /**
36
+ * Resolve audio input to base64 + format, supporting:
37
+ * - data: URLs (pass through)
38
+ * - http(s) URLs (fetch)
39
+ * - local file paths (sandboxed read)
40
+ */
41
+ async function resolveAudioInput(audioPath) {
42
+ const trimmed = audioPath.trim();
43
+ if (!trimmed)
44
+ throw new Error('audio_path is empty');
45
+ // Data URL
46
+ if (trimmed.startsWith('data:')) {
47
+ const match = trimmed.match(/^data:audio\/([^;,]+)(?:;[^,]*)*;base64,(.+)$/);
48
+ if (!match)
49
+ throw new Error('Invalid audio data URL format');
50
+ return { data: match[2], format: match[1] };
51
+ }
52
+ // HTTP URL
53
+ if (/^https?:\/\//i.test(trimmed)) {
54
+ const { fetchHttpResource } = await import('./fetch-utils.js');
55
+ const { buffer, contentType } = await fetchHttpResource(trimmed, {
56
+ timeoutMs: 60_000,
57
+ maxBytes: 100 * 1024 * 1024,
58
+ maxRedirects: 8,
59
+ });
60
+ const format = contentType?.match(/audio\/(\w+)/)?.[1] || 'mp3';
61
+ return { data: buffer.toString('base64'), format };
62
+ }
63
+ // Local file
64
+ const abs = await resolveSafeInputPath(trimmed);
65
+ const buf = await fs.readFile(abs);
66
+ const ext = path.extname(abs);
67
+ const format = audioFormatFromExt(ext);
68
+ return { data: buf.toString('base64'), format };
69
+ }
70
+ export async function handleSpeechToText(request, apiClient) {
71
+ const args = request.params.arguments ?? {};
72
+ const { audio_path, model, language, response_format, temperature, cache, cache_ttl, cache_clear, } = args;
73
+ if (!audio_path?.trim()) {
74
+ return toolError(ErrorCode.INVALID_INPUT, 'audio_path is required.');
75
+ }
76
+ if (response_format && !VALID_RESPONSE_FORMATS.has(response_format)) {
77
+ return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_RESPONSE_FORMATS].join(', ')}.`);
78
+ }
79
+ logger.audit('speech_to_text.start', {
80
+ model: model || DEFAULT_MODEL,
81
+ audio_path: audio_path.startsWith('data:') ? 'data_url' : audio_path.slice(0, 80),
82
+ language,
83
+ response_format,
84
+ });
85
+ // Resolve audio input
86
+ let audioInput;
87
+ try {
88
+ audioInput = await resolveAudioInput(audio_path);
89
+ }
90
+ catch (err) {
91
+ if (err instanceof UnsafeOutputPathError)
92
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
93
+ const msg = err instanceof Error ? err.message : String(err);
94
+ if (msg.includes('Blocked host'))
95
+ return toolErrorFrom(ErrorCode.UPSTREAM_REFUSED, err);
96
+ return toolErrorFrom(ErrorCode.INVALID_INPUT, err);
97
+ }
98
+ // Build request body
99
+ const body = {
100
+ model: model || DEFAULT_MODEL,
101
+ input_audio: {
102
+ data: audioInput.data,
103
+ format: audioInput.format,
104
+ },
105
+ };
106
+ if (language)
107
+ body.language = language;
108
+ if (response_format)
109
+ body.response_format = response_format;
110
+ if (typeof temperature === 'number')
111
+ body.temperature = temperature;
112
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
113
+ let response;
114
+ try {
115
+ response = await apiClient.transcribeAudio(body, headers);
116
+ }
117
+ catch (err) {
118
+ return classifyUpstreamError(err, 'speech_to_text');
119
+ }
120
+ const text = response.text;
121
+ if (!text) {
122
+ return toolError(ErrorCode.INTERNAL, 'Transcription returned no text.', {
123
+ response_keys: Object.keys(response),
124
+ });
125
+ }
126
+ const baseMeta = {
127
+ server_version: SERVER_VERSION,
128
+ model: model || DEFAULT_MODEL,
129
+ };
130
+ if (response.language)
131
+ baseMeta.language = response.language;
132
+ if (response.duration)
133
+ baseMeta.duration_seconds = response.duration;
134
+ if (response.usage)
135
+ baseMeta.usage = response.usage;
136
+ return {
137
+ content: [{ type: 'text', text }],
138
+ _meta: baseMeta,
139
+ };
140
+ }
@@ -11,3 +11,11 @@ export interface StructuredResult<T = unknown> {
11
11
  * format. `meta` is merged on top of the default `server_version` stamp.
12
12
  */
13
13
  export declare function buildStructuredResult<T>(data: T, meta?: Record<string, unknown>): StructuredResult<T>;
14
+ /** Read typed JSON from an MCP tool result (structuredContent or legacy text). */
15
+ export declare function readToolPayload<T = unknown>(result: {
16
+ structuredContent?: T;
17
+ content?: Array<{
18
+ type: string;
19
+ text?: string;
20
+ }>;
21
+ }): T;
@@ -22,3 +22,14 @@ export function buildStructuredResult(data, meta = {}) {
22
22
  _meta: { server_version: SERVER_VERSION, ...meta },
23
23
  };
24
24
  }
25
+ /** Read typed JSON from an MCP tool result (structuredContent or legacy text). */
26
+ export function readToolPayload(result) {
27
+ if (result.structuredContent !== undefined) {
28
+ return result.structuredContent;
29
+ }
30
+ const text = result.content?.[0]?.text;
31
+ if (text === undefined) {
32
+ throw new Error('tool result has no structuredContent or content text');
33
+ }
34
+ return JSON.parse(text);
35
+ }
@@ -0,0 +1,29 @@
1
+ import type { OpenRouterAPIClient } from '../openrouter-api.js';
2
+ import { type CacheOptions } from './cache.js';
3
+ export interface TextToSpeechRequest extends CacheOptions {
4
+ input: string;
5
+ model?: string;
6
+ voice?: string;
7
+ response_format?: string;
8
+ speed?: number;
9
+ instructions?: string;
10
+ save_path?: string;
11
+ }
12
+ export declare function handleTextToSpeech(request: {
13
+ params: {
14
+ arguments: TextToSpeechRequest;
15
+ };
16
+ }, apiClient: OpenRouterAPIClient): Promise<import("../errors.js").ToolErrorResult | {
17
+ content: ({
18
+ type: "text";
19
+ text: string;
20
+ mimeType?: undefined;
21
+ data?: undefined;
22
+ } | {
23
+ type: "audio";
24
+ mimeType: string;
25
+ data: string;
26
+ text?: undefined;
27
+ })[];
28
+ _meta: Record<string, unknown>;
29
+ }>;
@@ -0,0 +1,105 @@
1
+ /**
2
+ * text_to_speech — uses OpenRouter's dedicated POST /api/v1/audio/speech
3
+ * endpoint (launched May 2026) for text-to-speech. Faster and more cost-efficient
4
+ * than routing through chat completions with audio modality.
5
+ *
6
+ * Supported providers: OpenAI (GPT-4o Mini TTS), Google (Gemini Flash TTS),
7
+ * Mistral (Voxtral Mini TTS).
8
+ */
9
+ import { promises as fs } from 'fs';
10
+ import { extname } from 'path';
11
+ import { resolveSafeOutputPath, UnsafeOutputPathError } from './path-safety.js';
12
+ import { ErrorCode, toolError, toolErrorFrom } from '../errors.js';
13
+ import { SERVER_VERSION } from '../version.js';
14
+ import { logger } from '../logger.js';
15
+ import { classifyUpstreamError } from './openrouter-errors.js';
16
+ import { buildCacheHeaders } from './cache.js';
17
+ const DEFAULT_MODEL = 'openai/gpt-4o-mini-tts-2025-12-15';
18
+ const DEFAULT_VOICE = 'alloy';
19
+ const VALID_FORMATS = new Set(['mp3', 'opus', 'aac', 'flac', 'wav', 'pcm']);
20
+ export async function handleTextToSpeech(request, apiClient) {
21
+ const args = request.params.arguments ?? {};
22
+ const { input, model, voice, response_format, speed, instructions, save_path, cache, cache_ttl, cache_clear, } = args;
23
+ if (!input?.trim()) {
24
+ return toolError(ErrorCode.INVALID_INPUT, 'input text is required.');
25
+ }
26
+ if (response_format && !VALID_FORMATS.has(response_format)) {
27
+ return toolError(ErrorCode.INVALID_INPUT, `response_format '${response_format}' is not supported. Valid: ${[...VALID_FORMATS].join(', ')}.`);
28
+ }
29
+ logger.audit('text_to_speech.start', {
30
+ model: model || DEFAULT_MODEL,
31
+ voice: voice || DEFAULT_VOICE,
32
+ response_format: response_format || 'mp3',
33
+ input_preview: input.slice(0, 80),
34
+ save_path: save_path ? 'provided' : 'none',
35
+ });
36
+ // Resolve save path early
37
+ let safeSavePath = null;
38
+ if (save_path) {
39
+ try {
40
+ safeSavePath = await resolveSafeOutputPath(save_path);
41
+ }
42
+ catch (err) {
43
+ if (err instanceof UnsafeOutputPathError)
44
+ return toolErrorFrom(ErrorCode.UNSAFE_PATH, err);
45
+ return toolErrorFrom(ErrorCode.INTERNAL, err);
46
+ }
47
+ }
48
+ // Build request body
49
+ const body = {
50
+ model: model || DEFAULT_MODEL,
51
+ input,
52
+ voice: voice || DEFAULT_VOICE,
53
+ };
54
+ if (response_format)
55
+ body.response_format = response_format;
56
+ if (typeof speed === 'number' && speed > 0)
57
+ body.speed = speed;
58
+ if (instructions)
59
+ body.instructions = instructions;
60
+ const headers = buildCacheHeaders({ cache, cache_ttl, cache_clear });
61
+ let result;
62
+ try {
63
+ result = await apiClient.generateSpeech(body, headers);
64
+ }
65
+ catch (err) {
66
+ return classifyUpstreamError(err, 'text_to_speech');
67
+ }
68
+ const { buffer, contentType } = result;
69
+ const mimeType = contentType.split(';')[0]?.trim() || 'audio/mpeg';
70
+ // Determine file extension from format
71
+ const ext = response_format || 'mp3';
72
+ const baseMeta = {
73
+ server_version: SERVER_VERSION,
74
+ model: model || DEFAULT_MODEL,
75
+ mime: mimeType,
76
+ size_bytes: buffer.length,
77
+ voice: voice || DEFAULT_VOICE,
78
+ };
79
+ if (safeSavePath) {
80
+ // Ensure extension matches
81
+ const currentExt = extname(safeSavePath).toLowerCase().slice(1);
82
+ const actualPath = currentExt === ext ? safeSavePath : `${safeSavePath}.${ext}`;
83
+ try {
84
+ await fs.writeFile(actualPath, buffer);
85
+ }
86
+ catch (err) {
87
+ return toolErrorFrom(ErrorCode.INTERNAL, err, 'Write');
88
+ }
89
+ baseMeta.save_path = actualPath;
90
+ return {
91
+ content: [
92
+ { type: 'text', text: `Speech saved to: ${actualPath}` },
93
+ { type: 'audio', mimeType, data: buffer.toString('base64') },
94
+ ],
95
+ _meta: baseMeta,
96
+ };
97
+ }
98
+ return {
99
+ content: [
100
+ { type: 'text', text: `Speech generated (${buffer.length} bytes, ${mimeType}).` },
101
+ { type: 'audio', mimeType, data: buffer.toString('base64') },
102
+ ],
103
+ _meta: baseMeta,
104
+ };
105
+ }
@@ -9,6 +9,7 @@
9
9
  import path from 'node:path';
10
10
  import { promises as fs } from 'node:fs';
11
11
  import { readEnvInt, fetchHttpResource, parseBase64DataUrl } from './fetch-utils.js';
12
+ import { resolveSafeInputPath } from './path-safety.js';
12
13
  export { isBlockedIPv4, assertUrlSafeForFetch } from './fetch-utils.js';
13
14
  const DEFAULT_FETCH_TIMEOUT_MS = 60_000;
14
15
  const DEFAULT_MAX_DOWNLOAD_BYTES = 100 * 1024 * 1024; // 100 MB
@@ -93,10 +94,7 @@ export function detectVideoFormat(buffer) {
93
94
  }
94
95
  if (buffer.length >= 4) {
95
96
  // EBML header — WebM & Matroska.
96
- if (buffer[0] === 0x1a &&
97
- buffer[1] === 0x45 &&
98
- buffer[2] === 0xdf &&
99
- buffer[3] === 0xa3) {
97
+ if (buffer[0] === 0x1a && buffer[1] === 0x45 && buffer[2] === 0xdf && buffer[3] === 0xa3) {
100
98
  return 'webm';
101
99
  }
102
100
  // MPEG-PS / MPEG-TS start codes.
@@ -146,9 +144,7 @@ export async function prepareVideoData(source) {
146
144
  maxRedirects: getMaxRedirects(),
147
145
  });
148
146
  const urlPath = new URL(source).pathname;
149
- const format = detectVideoFormat(buffer) ??
150
- getVideoFormat(urlPath) ??
151
- formatFromContentType(contentType);
147
+ const format = detectVideoFormat(buffer) ?? getVideoFormat(urlPath) ?? formatFromContentType(contentType);
152
148
  if (!format) {
153
149
  throw new Error(`Could not determine video format from ${source}. Supported: ${SUPPORTED_VIDEO_FORMATS.join(', ')}`);
154
150
  }
@@ -160,8 +156,9 @@ export async function prepareVideoData(source) {
160
156
  };
161
157
  }
162
158
  // --- local file ---
163
- const buffer = await fs.readFile(source);
164
- const format = detectVideoFormat(buffer) ?? getVideoFormat(source);
159
+ const safe = await resolveSafeInputPath(source);
160
+ const buffer = await fs.readFile(safe);
161
+ const format = detectVideoFormat(buffer) ?? getVideoFormat(safe);
165
162
  if (!format) {
166
163
  throw new Error(`Unsupported video format for file: ${source}. Supported: ${SUPPORTED_VIDEO_FORMATS.join(', ')}`);
167
164
  }