@framers/agentos-ext-voice-synthesis 2.0.1 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +96 -21
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/tools/speechToText.d.ts.map +1 -1
- package/dist/tools/speechToText.js +1 -1
- package/dist/tools/speechToText.js.map +1 -1
- package/dist/tools/textToSpeech.d.ts.map +1 -1
- package/dist/tools/textToSpeech.js +1 -0
- package/dist/tools/textToSpeech.js.map +1 -1
- package/manifest.json +1 -1
- package/package.json +16 -8
- package/src/index.ts +0 -92
- package/src/tools/speechToText.ts +0 -650
- package/src/tools/textToSpeech.ts +0 -287
- package/test/textToSpeech.spec.ts +0 -398
- package/tsconfig.json +0 -22
- package/vitest.config.ts +0 -10
|
@@ -1,287 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Multi-provider TTS Tool — text-to-speech synthesis.
|
|
3
|
-
*
|
|
4
|
-
* Supports: OpenAI TTS, ElevenLabs, Ollama (local), any OpenAI-compatible TTS API.
|
|
5
|
-
* Auto-detects available provider from API keys in environment.
|
|
6
|
-
*/
|
|
7
|
-
|
|
8
|
-
import type { ITool, ToolExecutionContext, ToolExecutionResult, JSONSchemaObject } from '@framers/agentos';
|
|
9
|
-
|
|
10
|
-
export type TTSProvider = 'openai' | 'elevenlabs' | 'ollama' | 'auto';
|
|
11
|
-
|
|
12
|
-
export interface TTSInput {
|
|
13
|
-
text: string;
|
|
14
|
-
voice?: string;
|
|
15
|
-
model?: string;
|
|
16
|
-
provider?: TTSProvider;
|
|
17
|
-
/** ElevenLabs-specific */
|
|
18
|
-
stability?: number;
|
|
19
|
-
/** ElevenLabs-specific */
|
|
20
|
-
similarity_boost?: number;
|
|
21
|
-
/** OpenAI-specific: speed 0.25-4.0 */
|
|
22
|
-
speed?: number;
|
|
23
|
-
/** Output format: mp3, opus, aac, flac, wav */
|
|
24
|
-
format?: string;
|
|
25
|
-
}
|
|
26
|
-
|
|
27
|
-
export interface TTSOutput {
|
|
28
|
-
text: string;
|
|
29
|
-
voice: string;
|
|
30
|
-
model: string;
|
|
31
|
-
provider: string;
|
|
32
|
-
audioBase64: string;
|
|
33
|
-
contentType: string;
|
|
34
|
-
durationEstimateMs: number;
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
// ── ElevenLabs voice name → ID mapping ──
|
|
38
|
-
const ELEVENLABS_VOICES: Record<string, string> = {
|
|
39
|
-
rachel: '21m00Tcm4TlvDq8ikWAM',
|
|
40
|
-
domi: 'AZnzlk1XvdvUeBnXmlld',
|
|
41
|
-
bella: 'EXAVITQu4vr4xnSDxMaL',
|
|
42
|
-
antoni: 'ErXwobaYiN019PkySvjV',
|
|
43
|
-
josh: 'TxGEqnHWrfWFTfGW9XjX',
|
|
44
|
-
arnold: 'VR6AewLTigWG4xSOukaG',
|
|
45
|
-
adam: 'pNInz6obpgDQGcFmaJgB',
|
|
46
|
-
sam: 'yoZ06aMxZJJ28mfd3POQ',
|
|
47
|
-
};
|
|
48
|
-
|
|
49
|
-
// ── OpenAI voice options ──
|
|
50
|
-
const OPENAI_VOICES = ['alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer'];
|
|
51
|
-
|
|
52
|
-
export interface TTSConfig {
|
|
53
|
-
openaiApiKey?: string;
|
|
54
|
-
openaiBaseUrl?: string;
|
|
55
|
-
elevenLabsApiKey?: string;
|
|
56
|
-
ollamaBaseUrl?: string;
|
|
57
|
-
defaultProvider?: TTSProvider;
|
|
58
|
-
}
|
|
59
|
-
|
|
60
|
-
export class TextToSpeechTool implements ITool<TTSInput, TTSOutput> {
|
|
61
|
-
readonly id = 'tts-multi-provider-v1';
|
|
62
|
-
readonly name = 'text_to_speech';
|
|
63
|
-
readonly displayName = 'Text to Speech';
|
|
64
|
-
readonly description =
|
|
65
|
-
'Convert text to speech audio. Supports multiple providers: OpenAI TTS (alloy/echo/fable/onyx/nova/shimmer), ' +
|
|
66
|
-
'ElevenLabs (rachel/domi/bella/antoni/josh/arnold/adam/sam), or local Ollama TTS. ' +
|
|
67
|
-
'Auto-detects available provider from API keys. Returns base64-encoded audio.';
|
|
68
|
-
readonly category = 'media';
|
|
69
|
-
readonly version = '2.0.0';
|
|
70
|
-
readonly hasSideEffects = false;
|
|
71
|
-
|
|
72
|
-
readonly inputSchema: JSONSchemaObject = {
|
|
73
|
-
type: 'object',
|
|
74
|
-
properties: {
|
|
75
|
-
text: { type: 'string', description: 'Text to convert to speech. Max 5000 chars.' },
|
|
76
|
-
voice: {
|
|
77
|
-
type: 'string',
|
|
78
|
-
description:
|
|
79
|
-
'Voice name. OpenAI: alloy, echo, fable, onyx, nova (default), shimmer. ' +
|
|
80
|
-
'ElevenLabs: rachel (default), domi, bella, antoni, josh, arnold, adam, sam. ' +
|
|
81
|
-
'Or a custom voice ID.',
|
|
82
|
-
},
|
|
83
|
-
model: {
|
|
84
|
-
type: 'string',
|
|
85
|
-
description: 'TTS model. OpenAI: tts-1 (default), tts-1-hd. ElevenLabs: eleven_monolingual_v1 (default), eleven_multilingual_v2.',
|
|
86
|
-
},
|
|
87
|
-
provider: {
|
|
88
|
-
type: 'string',
|
|
89
|
-
enum: ['openai', 'elevenlabs', 'ollama', 'auto'],
|
|
90
|
-
description: 'TTS provider. Default: auto (detects from available API keys).',
|
|
91
|
-
},
|
|
92
|
-
speed: { type: 'number', minimum: 0.25, maximum: 4.0, description: 'OpenAI speed multiplier (0.25-4.0).' },
|
|
93
|
-
stability: { type: 'number', minimum: 0, maximum: 1, description: 'ElevenLabs voice stability (0-1).' },
|
|
94
|
-
similarity_boost: { type: 'number', minimum: 0, maximum: 1, description: 'ElevenLabs similarity boost (0-1).' },
|
|
95
|
-
format: { type: 'string', enum: ['mp3', 'opus', 'aac', 'flac', 'wav'], description: 'Output audio format.' },
|
|
96
|
-
},
|
|
97
|
-
required: ['text'],
|
|
98
|
-
};
|
|
99
|
-
|
|
100
|
-
readonly requiredCapabilities = ['capability:tts'];
|
|
101
|
-
|
|
102
|
-
private config: TTSConfig;
|
|
103
|
-
|
|
104
|
-
constructor(config?: TTSConfig) {
|
|
105
|
-
this.config = {
|
|
106
|
-
openaiApiKey: config?.openaiApiKey || process.env.OPENAI_API_KEY || '',
|
|
107
|
-
openaiBaseUrl: config?.openaiBaseUrl || process.env.OPENAI_BASE_URL || 'https://api.openai.com/v1',
|
|
108
|
-
elevenLabsApiKey: config?.elevenLabsApiKey || process.env.ELEVENLABS_API_KEY || '',
|
|
109
|
-
ollamaBaseUrl: config?.ollamaBaseUrl || process.env.OLLAMA_BASE_URL || 'http://localhost:11434',
|
|
110
|
-
defaultProvider: config?.defaultProvider || (process.env.TTS_PROVIDER as TTSProvider) || 'auto',
|
|
111
|
-
};
|
|
112
|
-
}
|
|
113
|
-
|
|
114
|
-
private resolveProvider(requested?: TTSProvider): TTSProvider | null {
|
|
115
|
-
const pref = requested || this.config.defaultProvider || 'auto';
|
|
116
|
-
if (pref !== 'auto') {
|
|
117
|
-
// Verify the requested provider has credentials
|
|
118
|
-
if (pref === 'openai' && this.config.openaiApiKey) return 'openai';
|
|
119
|
-
if (pref === 'elevenlabs' && this.config.elevenLabsApiKey) return 'elevenlabs';
|
|
120
|
-
if (pref === 'ollama') return 'ollama';
|
|
121
|
-
// Fall through to auto if requested provider isn't configured
|
|
122
|
-
}
|
|
123
|
-
|
|
124
|
-
// Auto-detect: prefer OpenAI (cheaper, faster), then ElevenLabs, then Ollama
|
|
125
|
-
if (this.config.openaiApiKey) return 'openai';
|
|
126
|
-
if (this.config.elevenLabsApiKey) return 'elevenlabs';
|
|
127
|
-
return 'ollama'; // Local fallback — may or may not have TTS model
|
|
128
|
-
}
|
|
129
|
-
|
|
130
|
-
async execute(args: TTSInput, _context: ToolExecutionContext): Promise<ToolExecutionResult<TTSOutput>> {
|
|
131
|
-
const text = args.text.slice(0, 5000);
|
|
132
|
-
const provider = this.resolveProvider(args.provider);
|
|
133
|
-
|
|
134
|
-
if (!provider) {
|
|
135
|
-
return {
|
|
136
|
-
success: false,
|
|
137
|
-
error:
|
|
138
|
-
'No TTS provider available. Set one of: OPENAI_API_KEY, ELEVENLABS_API_KEY, or configure Ollama with a TTS model. ' +
|
|
139
|
-
'Get an OpenAI key at https://platform.openai.com/api-keys or ElevenLabs at https://elevenlabs.io',
|
|
140
|
-
};
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
try {
|
|
144
|
-
switch (provider) {
|
|
145
|
-
case 'openai':
|
|
146
|
-
return await this.synthesizeOpenAI(text, args);
|
|
147
|
-
case 'elevenlabs':
|
|
148
|
-
return await this.synthesizeElevenLabs(text, args);
|
|
149
|
-
case 'ollama':
|
|
150
|
-
return await this.synthesizeOllama(text, args);
|
|
151
|
-
default:
|
|
152
|
-
return { success: false, error: `Unknown TTS provider: ${provider}` };
|
|
153
|
-
}
|
|
154
|
-
} catch (err: any) {
|
|
155
|
-
return { success: false, error: `TTS failed (${provider}): ${err.message}` };
|
|
156
|
-
}
|
|
157
|
-
}
|
|
158
|
-
|
|
159
|
-
// ── OpenAI TTS ──
|
|
160
|
-
|
|
161
|
-
private async synthesizeOpenAI(text: string, args: TTSInput): Promise<ToolExecutionResult<TTSOutput>> {
|
|
162
|
-
const voice = args.voice && OPENAI_VOICES.includes(args.voice) ? args.voice : 'nova';
|
|
163
|
-
const model = args.model || 'tts-1';
|
|
164
|
-
const format = args.format || 'mp3';
|
|
165
|
-
|
|
166
|
-
const response = await fetch(`${this.config.openaiBaseUrl}/audio/speech`, {
|
|
167
|
-
method: 'POST',
|
|
168
|
-
headers: {
|
|
169
|
-
Authorization: `Bearer ${this.config.openaiApiKey}`,
|
|
170
|
-
'Content-Type': 'application/json',
|
|
171
|
-
},
|
|
172
|
-
body: JSON.stringify({
|
|
173
|
-
model,
|
|
174
|
-
voice,
|
|
175
|
-
input: text,
|
|
176
|
-
response_format: format,
|
|
177
|
-
speed: args.speed,
|
|
178
|
-
}),
|
|
179
|
-
});
|
|
180
|
-
|
|
181
|
-
if (!response.ok) {
|
|
182
|
-
const err = await response.text();
|
|
183
|
-
return { success: false, error: `OpenAI TTS error (${response.status}): ${err.slice(0, 300)}` };
|
|
184
|
-
}
|
|
185
|
-
|
|
186
|
-
const buf = await response.arrayBuffer();
|
|
187
|
-
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
188
|
-
const contentType = format === 'opus' ? 'audio/opus' : format === 'wav' ? 'audio/wav' : 'audio/mpeg';
|
|
189
|
-
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
190
|
-
|
|
191
|
-
return {
|
|
192
|
-
success: true,
|
|
193
|
-
output: { text, voice, model, provider: 'openai', audioBase64, contentType, durationEstimateMs },
|
|
194
|
-
contentType,
|
|
195
|
-
};
|
|
196
|
-
}
|
|
197
|
-
|
|
198
|
-
// ── ElevenLabs TTS ──
|
|
199
|
-
|
|
200
|
-
private async synthesizeElevenLabs(text: string, args: TTSInput): Promise<ToolExecutionResult<TTSOutput>> {
|
|
201
|
-
const voiceId = ELEVENLABS_VOICES[(args.voice || 'rachel').toLowerCase()] || args.voice || ELEVENLABS_VOICES.rachel;
|
|
202
|
-
const model = args.model || 'eleven_monolingual_v1';
|
|
203
|
-
|
|
204
|
-
const response = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`, {
|
|
205
|
-
method: 'POST',
|
|
206
|
-
headers: {
|
|
207
|
-
'xi-api-key': this.config.elevenLabsApiKey!,
|
|
208
|
-
'Content-Type': 'application/json',
|
|
209
|
-
Accept: 'audio/mpeg',
|
|
210
|
-
},
|
|
211
|
-
body: JSON.stringify({
|
|
212
|
-
text,
|
|
213
|
-
model_id: model,
|
|
214
|
-
voice_settings: {
|
|
215
|
-
stability: args.stability ?? 0.5,
|
|
216
|
-
similarity_boost: args.similarity_boost ?? 0.75,
|
|
217
|
-
},
|
|
218
|
-
}),
|
|
219
|
-
});
|
|
220
|
-
|
|
221
|
-
if (!response.ok) {
|
|
222
|
-
const err = await response.text();
|
|
223
|
-
return { success: false, error: `ElevenLabs error (${response.status}): ${err.slice(0, 300)}` };
|
|
224
|
-
}
|
|
225
|
-
|
|
226
|
-
const buf = await response.arrayBuffer();
|
|
227
|
-
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
228
|
-
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
229
|
-
|
|
230
|
-
return {
|
|
231
|
-
success: true,
|
|
232
|
-
output: {
|
|
233
|
-
text,
|
|
234
|
-
voice: args.voice || 'rachel',
|
|
235
|
-
model,
|
|
236
|
-
provider: 'elevenlabs',
|
|
237
|
-
audioBase64,
|
|
238
|
-
contentType: 'audio/mpeg',
|
|
239
|
-
durationEstimateMs,
|
|
240
|
-
},
|
|
241
|
-
contentType: 'audio/mpeg',
|
|
242
|
-
};
|
|
243
|
-
}
|
|
244
|
-
|
|
245
|
-
// ── Ollama TTS (local, experimental) ──
|
|
246
|
-
|
|
247
|
-
private async synthesizeOllama(text: string, args: TTSInput): Promise<ToolExecutionResult<TTSOutput>> {
|
|
248
|
-
// Ollama doesn't natively support TTS yet, but some models (e.g., bark, piper)
|
|
249
|
-
// can be served via OpenAI-compatible endpoints. Try the OpenAI-compat path.
|
|
250
|
-
const voice = args.voice || 'default';
|
|
251
|
-
const model = args.model || 'tts'; // User must have a TTS model loaded
|
|
252
|
-
|
|
253
|
-
try {
|
|
254
|
-
const response = await fetch(`${this.config.ollamaBaseUrl}/v1/audio/speech`, {
|
|
255
|
-
method: 'POST',
|
|
256
|
-
headers: { 'Content-Type': 'application/json' },
|
|
257
|
-
body: JSON.stringify({ model, voice, input: text }),
|
|
258
|
-
});
|
|
259
|
-
|
|
260
|
-
if (!response.ok) {
|
|
261
|
-
return {
|
|
262
|
-
success: false,
|
|
263
|
-
error:
|
|
264
|
-
`Ollama TTS not available (${response.status}). Ollama doesn't natively support TTS yet. ` +
|
|
265
|
-
'Set OPENAI_API_KEY or ELEVENLABS_API_KEY for cloud TTS, or use a dedicated local TTS server.',
|
|
266
|
-
};
|
|
267
|
-
}
|
|
268
|
-
|
|
269
|
-
const buf = await response.arrayBuffer();
|
|
270
|
-
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
271
|
-
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
272
|
-
|
|
273
|
-
return {
|
|
274
|
-
success: true,
|
|
275
|
-
output: { text, voice, model, provider: 'ollama', audioBase64, contentType: 'audio/mpeg', durationEstimateMs },
|
|
276
|
-
contentType: 'audio/mpeg',
|
|
277
|
-
};
|
|
278
|
-
} catch {
|
|
279
|
-
return {
|
|
280
|
-
success: false,
|
|
281
|
-
error:
|
|
282
|
-
'Ollama TTS endpoint not reachable. Ollama doesn\'t natively support TTS yet. ' +
|
|
283
|
-
'Set OPENAI_API_KEY or ELEVENLABS_API_KEY for cloud TTS.',
|
|
284
|
-
};
|
|
285
|
-
}
|
|
286
|
-
}
|
|
287
|
-
}
|
|
@@ -1,398 +0,0 @@
|
|
|
1
|
-
import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest';
|
|
2
|
-
|
|
3
|
-
const mockFetch = vi.fn();
|
|
4
|
-
vi.stubGlobal('fetch', mockFetch);
|
|
5
|
-
|
|
6
|
-
// Clear env vars that affect provider detection
|
|
7
|
-
const savedEnv: Record<string, string | undefined> = {};
|
|
8
|
-
const envKeys = [
|
|
9
|
-
'OPENAI_API_KEY',
|
|
10
|
-
'ELEVENLABS_API_KEY',
|
|
11
|
-
'OPENAI_BASE_URL',
|
|
12
|
-
'OLLAMA_BASE_URL',
|
|
13
|
-
'TTS_PROVIDER',
|
|
14
|
-
'DEEPGRAM_API_KEY',
|
|
15
|
-
'DEEPGRAM_BASE_URL',
|
|
16
|
-
'WHISPER_LOCAL_BASE_URL',
|
|
17
|
-
'STT_PROVIDER',
|
|
18
|
-
];
|
|
19
|
-
|
|
20
|
-
const { TextToSpeechTool } = await import('../src/tools/textToSpeech.js');
|
|
21
|
-
const { SpeechToTextTool } = await import('../src/tools/speechToText.js');
|
|
22
|
-
const { createExtensionPack } = await import('../src/index.js');
|
|
23
|
-
|
|
24
|
-
describe('TextToSpeechTool', () => {
|
|
25
|
-
const ctx = {} as any;
|
|
26
|
-
|
|
27
|
-
beforeEach(() => {
|
|
28
|
-
vi.clearAllMocks();
|
|
29
|
-
// Save and clear env
|
|
30
|
-
for (const key of envKeys) {
|
|
31
|
-
savedEnv[key] = process.env[key];
|
|
32
|
-
delete process.env[key];
|
|
33
|
-
}
|
|
34
|
-
});
|
|
35
|
-
|
|
36
|
-
afterEach(() => {
|
|
37
|
-
// Restore env
|
|
38
|
-
for (const key of envKeys) {
|
|
39
|
-
if (savedEnv[key] !== undefined) process.env[key] = savedEnv[key];
|
|
40
|
-
else delete process.env[key];
|
|
41
|
-
}
|
|
42
|
-
});
|
|
43
|
-
|
|
44
|
-
describe('metadata', () => {
|
|
45
|
-
it('has correct id and name', () => {
|
|
46
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test' });
|
|
47
|
-
expect(tool.id).toBe('tts-multi-provider-v1');
|
|
48
|
-
expect(tool.name).toBe('text_to_speech');
|
|
49
|
-
});
|
|
50
|
-
|
|
51
|
-
it('has valid input schema with text required', () => {
|
|
52
|
-
const tool = new TextToSpeechTool({});
|
|
53
|
-
expect(tool.inputSchema.type).toBe('object');
|
|
54
|
-
expect(tool.inputSchema.required).toContain('text');
|
|
55
|
-
});
|
|
56
|
-
|
|
57
|
-
it('has no side effects', () => {
|
|
58
|
-
const tool = new TextToSpeechTool({});
|
|
59
|
-
expect(tool.hasSideEffects).toBe(false);
|
|
60
|
-
});
|
|
61
|
-
|
|
62
|
-
it('describes multiple providers', () => {
|
|
63
|
-
const tool = new TextToSpeechTool({});
|
|
64
|
-
expect(tool.description).toContain('OpenAI');
|
|
65
|
-
expect(tool.description).toContain('ElevenLabs');
|
|
66
|
-
});
|
|
67
|
-
});
|
|
68
|
-
|
|
69
|
-
describe('provider resolution', () => {
|
|
70
|
-
it('falls back to Ollama when no API keys set', async () => {
|
|
71
|
-
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: '' });
|
|
72
|
-
mockFetch.mockRejectedValueOnce(new Error('ECONNREFUSED'));
|
|
73
|
-
const result = await tool.execute({ text: 'Hello' }, ctx);
|
|
74
|
-
expect(result.success).toBe(false);
|
|
75
|
-
expect(result.error).toContain('Ollama');
|
|
76
|
-
});
|
|
77
|
-
|
|
78
|
-
it('auto-detects OpenAI when key provided', async () => {
|
|
79
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
80
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
81
|
-
const result = await tool.execute({ text: 'Hello' }, ctx);
|
|
82
|
-
expect(result.success).toBe(true);
|
|
83
|
-
expect(result.output!.provider).toBe('openai');
|
|
84
|
-
});
|
|
85
|
-
|
|
86
|
-
it('auto-detects ElevenLabs when only that key set', async () => {
|
|
87
|
-
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
88
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
89
|
-
const result = await tool.execute({ text: 'Hello' }, ctx);
|
|
90
|
-
expect(result.success).toBe(true);
|
|
91
|
-
expect(result.output!.provider).toBe('elevenlabs');
|
|
92
|
-
});
|
|
93
|
-
|
|
94
|
-
it('respects explicit provider override', async () => {
|
|
95
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: 'el-test' });
|
|
96
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
97
|
-
const result = await tool.execute({ text: 'Hello', provider: 'elevenlabs' }, ctx);
|
|
98
|
-
expect(result.success).toBe(true);
|
|
99
|
-
expect(result.output!.provider).toBe('elevenlabs');
|
|
100
|
-
});
|
|
101
|
-
});
|
|
102
|
-
|
|
103
|
-
describe('OpenAI TTS', () => {
|
|
104
|
-
it('synthesizes with default voice (nova)', async () => {
|
|
105
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
106
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
107
|
-
const result = await tool.execute({ text: 'Hello world' }, ctx);
|
|
108
|
-
expect(result.success).toBe(true);
|
|
109
|
-
expect(result.output!.voice).toBe('nova');
|
|
110
|
-
expect(result.output!.provider).toBe('openai');
|
|
111
|
-
expect(result.output!.contentType).toBe('audio/mpeg');
|
|
112
|
-
expect(result.output!.audioBase64).toBeTruthy();
|
|
113
|
-
});
|
|
114
|
-
|
|
115
|
-
it('sends correct request to OpenAI API', async () => {
|
|
116
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
117
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
118
|
-
await tool.execute({ text: 'Test', voice: 'shimmer', model: 'tts-1-hd' }, ctx);
|
|
119
|
-
const [url, opts] = mockFetch.mock.calls[0];
|
|
120
|
-
expect(url).toContain('/audio/speech');
|
|
121
|
-
const body = JSON.parse(opts.body);
|
|
122
|
-
expect(body.voice).toBe('shimmer');
|
|
123
|
-
expect(body.model).toBe('tts-1-hd');
|
|
124
|
-
});
|
|
125
|
-
|
|
126
|
-
it('handles API errors', async () => {
|
|
127
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
128
|
-
mockFetch.mockResolvedValueOnce({ ok: false, status: 401, text: async () => 'Unauthorized' });
|
|
129
|
-
const result = await tool.execute({ text: 'Test' }, ctx);
|
|
130
|
-
expect(result.success).toBe(false);
|
|
131
|
-
expect(result.error).toContain('401');
|
|
132
|
-
});
|
|
133
|
-
});
|
|
134
|
-
|
|
135
|
-
describe('ElevenLabs TTS', () => {
|
|
136
|
-
it('synthesizes with default voice (rachel)', async () => {
|
|
137
|
-
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
138
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
139
|
-
const result = await tool.execute({ text: 'Hello world' }, ctx);
|
|
140
|
-
expect(result.success).toBe(true);
|
|
141
|
-
expect(result.output!.voice).toBe('rachel');
|
|
142
|
-
expect(result.output!.provider).toBe('elevenlabs');
|
|
143
|
-
});
|
|
144
|
-
|
|
145
|
-
it('uses correct voice ID for named voice (josh)', async () => {
|
|
146
|
-
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
147
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
148
|
-
await tool.execute({ text: 'Test', voice: 'josh' }, ctx);
|
|
149
|
-
expect(mockFetch).toHaveBeenCalledWith(
|
|
150
|
-
expect.stringContaining('TxGEqnHWrfWFTfGW9XjX'),
|
|
151
|
-
expect.any(Object)
|
|
152
|
-
);
|
|
153
|
-
});
|
|
154
|
-
|
|
155
|
-
it('sends ElevenLabs API key header', async () => {
|
|
156
|
-
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
157
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
158
|
-
await tool.execute({ text: 'Test' }, ctx);
|
|
159
|
-
const [, opts] = mockFetch.mock.calls[0];
|
|
160
|
-
expect(opts.headers['xi-api-key']).toBe('el-test');
|
|
161
|
-
});
|
|
162
|
-
});
|
|
163
|
-
|
|
164
|
-
describe('common behavior', () => {
|
|
165
|
-
it('truncates text to 5000 chars', async () => {
|
|
166
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
167
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
168
|
-
const result = await tool.execute({ text: 'a'.repeat(6000) }, ctx);
|
|
169
|
-
expect(result.success).toBe(true);
|
|
170
|
-
expect(result.output!.text.length).toBe(5000);
|
|
171
|
-
});
|
|
172
|
-
|
|
173
|
-
it('estimates duration from word count', async () => {
|
|
174
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
175
|
-
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
176
|
-
const result = await tool.execute({ text: 'one two three four five' }, ctx);
|
|
177
|
-
expect(result.success).toBe(true);
|
|
178
|
-
expect(result.output!.durationEstimateMs).toBeGreaterThan(0);
|
|
179
|
-
});
|
|
180
|
-
|
|
181
|
-
it('handles network errors gracefully', async () => {
|
|
182
|
-
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
183
|
-
mockFetch.mockRejectedValueOnce(new Error('ECONNREFUSED'));
|
|
184
|
-
const result = await tool.execute({ text: 'Test' }, ctx);
|
|
185
|
-
expect(result.success).toBe(false);
|
|
186
|
-
expect(result.error).toContain('TTS failed');
|
|
187
|
-
});
|
|
188
|
-
});
|
|
189
|
-
});
|
|
190
|
-
|
|
191
|
-
describe('createExtensionPack', () => {
|
|
192
|
-
it('creates pack with correct metadata', () => {
|
|
193
|
-
const pack = createExtensionPack({ options: { elevenLabsApiKey: 'test' }, logger: { info: vi.fn() } });
|
|
194
|
-
expect(pack.name).toBe('@framers/agentos-ext-voice-synthesis');
|
|
195
|
-
expect(pack.version).toBe('2.0.0');
|
|
196
|
-
expect(pack.descriptors).toHaveLength(2);
|
|
197
|
-
expect(pack.descriptors[0].kind).toBe('tool');
|
|
198
|
-
expect(pack.descriptors[0].id).toBe('text_to_speech');
|
|
199
|
-
expect(pack.descriptors[1].id).toBe('speech_to_text');
|
|
200
|
-
});
|
|
201
|
-
});
|
|
202
|
-
|
|
203
|
-
describe('SpeechToTextTool', () => {
|
|
204
|
-
const ctx = {} as any;
|
|
205
|
-
|
|
206
|
-
beforeEach(() => {
|
|
207
|
-
vi.clearAllMocks();
|
|
208
|
-
for (const key of envKeys) {
|
|
209
|
-
savedEnv[key] = process.env[key];
|
|
210
|
-
delete process.env[key];
|
|
211
|
-
}
|
|
212
|
-
});
|
|
213
|
-
|
|
214
|
-
afterEach(() => {
|
|
215
|
-
for (const key of envKeys) {
|
|
216
|
-
if (savedEnv[key] !== undefined) process.env[key] = savedEnv[key];
|
|
217
|
-
else delete process.env[key];
|
|
218
|
-
}
|
|
219
|
-
});
|
|
220
|
-
|
|
221
|
-
it('has correct id and name', () => {
|
|
222
|
-
const tool = new SpeechToTextTool({ openaiApiKey: 'sk-test' });
|
|
223
|
-
expect(tool.id).toBe('stt-multi-provider-v1');
|
|
224
|
-
expect(tool.name).toBe('speech_to_text');
|
|
225
|
-
expect(tool.hasSideEffects).toBe(false);
|
|
226
|
-
});
|
|
227
|
-
|
|
228
|
-
it('fails when no provider is configured', async () => {
|
|
229
|
-
const tool = new SpeechToTextTool({ openaiApiKey: '' });
|
|
230
|
-
const result = await tool.execute({ audioBase64: Buffer.from('wav').toString('base64') }, ctx);
|
|
231
|
-
expect(result.success).toBe(false);
|
|
232
|
-
expect(result.error).toContain('OPENAI_API_KEY');
|
|
233
|
-
expect(result.error).toContain('DEEPGRAM_API_KEY');
|
|
234
|
-
});
|
|
235
|
-
|
|
236
|
-
it('fails when no audio input is provided', async () => {
|
|
237
|
-
const tool = new SpeechToTextTool({ openaiApiKey: 'sk-test' });
|
|
238
|
-
const result = await tool.execute({}, ctx);
|
|
239
|
-
expect(result.success).toBe(false);
|
|
240
|
-
expect(result.error).toContain('Provide either audioBase64 or audioUrl');
|
|
241
|
-
});
|
|
242
|
-
|
|
243
|
-
it('transcribes base64 audio through OpenAI Whisper', async () => {
|
|
244
|
-
const tool = new SpeechToTextTool({ openaiApiKey: 'sk-test' });
|
|
245
|
-
mockFetch.mockResolvedValueOnce({
|
|
246
|
-
ok: true,
|
|
247
|
-
headers: { get: vi.fn().mockReturnValue('application/json') },
|
|
248
|
-
json: async () => ({ text: 'hello world', language: 'en', duration: 1.5, segments: [{ text: 'hello world', start: 0, end: 1.5 }] }),
|
|
249
|
-
});
|
|
250
|
-
|
|
251
|
-
const result = await tool.execute(
|
|
252
|
-
{
|
|
253
|
-
audioBase64: 'data:audio/wav;base64,' + Buffer.from('fake-audio').toString('base64'),
|
|
254
|
-
responseFormat: 'verbose_json',
|
|
255
|
-
},
|
|
256
|
-
ctx,
|
|
257
|
-
);
|
|
258
|
-
|
|
259
|
-
expect(result.success).toBe(true);
|
|
260
|
-
expect(result.output!.text).toBe('hello world');
|
|
261
|
-
expect(result.output!.language).toBe('en');
|
|
262
|
-
expect(result.output!.provider).toBe('openai');
|
|
263
|
-
expect(result.output!.segments).toHaveLength(1);
|
|
264
|
-
const [url, opts] = mockFetch.mock.calls[0];
|
|
265
|
-
expect(url).toContain('/audio/transcriptions');
|
|
266
|
-
expect(opts.headers.Authorization).toBe('Bearer sk-test');
|
|
267
|
-
});
|
|
268
|
-
|
|
269
|
-
it('auto-detects Deepgram when OpenAI is unavailable', async () => {
|
|
270
|
-
const tool = new SpeechToTextTool({ openaiApiKey: '', deepgramApiKey: 'dg-test' });
|
|
271
|
-
mockFetch.mockResolvedValueOnce({
|
|
272
|
-
ok: true,
|
|
273
|
-
json: async () => ({
|
|
274
|
-
metadata: { duration: 2.1 },
|
|
275
|
-
results: {
|
|
276
|
-
utterances: [
|
|
277
|
-
{
|
|
278
|
-
transcript: 'hello from deepgram',
|
|
279
|
-
start: 0,
|
|
280
|
-
end: 2.1,
|
|
281
|
-
confidence: 0.98,
|
|
282
|
-
speaker: 0,
|
|
283
|
-
words: [{ word: 'hello', start: 0, end: 0.5, confidence: 0.9 }],
|
|
284
|
-
},
|
|
285
|
-
],
|
|
286
|
-
channels: [
|
|
287
|
-
{
|
|
288
|
-
alternatives: [
|
|
289
|
-
{
|
|
290
|
-
transcript: 'hello from deepgram',
|
|
291
|
-
confidence: 0.98,
|
|
292
|
-
detected_language: 'en',
|
|
293
|
-
},
|
|
294
|
-
],
|
|
295
|
-
},
|
|
296
|
-
],
|
|
297
|
-
},
|
|
298
|
-
}),
|
|
299
|
-
});
|
|
300
|
-
|
|
301
|
-
const result = await tool.execute(
|
|
302
|
-
{
|
|
303
|
-
audioBase64: Buffer.from('fake-audio').toString('base64'),
|
|
304
|
-
},
|
|
305
|
-
ctx,
|
|
306
|
-
);
|
|
307
|
-
|
|
308
|
-
expect(result.success).toBe(true);
|
|
309
|
-
expect(result.output!.provider).toBe('deepgram');
|
|
310
|
-
expect(result.output!.language).toBe('en');
|
|
311
|
-
expect(result.output!.segments).toHaveLength(1);
|
|
312
|
-
const [url, opts] = mockFetch.mock.calls[0];
|
|
313
|
-
expect(url).toContain('/listen?');
|
|
314
|
-
expect(opts.headers.Authorization).toBe('Token dg-test');
|
|
315
|
-
});
|
|
316
|
-
|
|
317
|
-
it('uses Whisper-local when explicitly requested', async () => {
|
|
318
|
-
const tool = new SpeechToTextTool({ whisperLocalBaseUrl: 'http://127.0.0.1:9000/v1' });
|
|
319
|
-
mockFetch.mockResolvedValueOnce({
|
|
320
|
-
ok: true,
|
|
321
|
-
headers: { get: vi.fn().mockReturnValue('application/json') },
|
|
322
|
-
json: async () => ({
|
|
323
|
-
text: 'local transcript',
|
|
324
|
-
language: 'en',
|
|
325
|
-
duration: 1.2,
|
|
326
|
-
}),
|
|
327
|
-
});
|
|
328
|
-
|
|
329
|
-
const result = await tool.execute(
|
|
330
|
-
{
|
|
331
|
-
provider: 'whisper-local',
|
|
332
|
-
audioBase64: Buffer.from('fake-audio').toString('base64'),
|
|
333
|
-
},
|
|
334
|
-
ctx,
|
|
335
|
-
);
|
|
336
|
-
|
|
337
|
-
expect(result.success).toBe(true);
|
|
338
|
-
expect(result.output!.provider).toBe('whisper-local');
|
|
339
|
-
expect(result.output!.text).toBe('local transcript');
|
|
340
|
-
expect(mockFetch.mock.calls[0][0]).toContain('127.0.0.1:9000/v1/audio/transcriptions');
|
|
341
|
-
});
|
|
342
|
-
|
|
343
|
-
it('downloads audio from a URL before transcribing', async () => {
|
|
344
|
-
const tool = new SpeechToTextTool({ openaiApiKey: 'sk-test' });
|
|
345
|
-
mockFetch
|
|
346
|
-
.mockResolvedValueOnce({
|
|
347
|
-
ok: true,
|
|
348
|
-
headers: { get: vi.fn().mockReturnValue('audio/mpeg') },
|
|
349
|
-
arrayBuffer: async () => new ArrayBuffer(12),
|
|
350
|
-
})
|
|
351
|
-
.mockResolvedValueOnce({
|
|
352
|
-
ok: true,
|
|
353
|
-
headers: { get: vi.fn().mockReturnValue('text/plain') },
|
|
354
|
-
text: async () => 'remote transcript',
|
|
355
|
-
});
|
|
356
|
-
|
|
357
|
-
const result = await tool.execute({ audioUrl: 'https://example.com/audio.mp3', responseFormat: 'text' }, ctx);
|
|
358
|
-
|
|
359
|
-
expect(result.success).toBe(true);
|
|
360
|
-
expect(result.output!.text).toBe('remote transcript');
|
|
361
|
-
expect(mockFetch).toHaveBeenNthCalledWith(1, 'https://example.com/audio.mp3');
|
|
362
|
-
expect(mockFetch.mock.calls[1][0]).toContain('/audio/transcriptions');
|
|
363
|
-
});
|
|
364
|
-
|
|
365
|
-
it('surfaces download failures clearly', async () => {
|
|
366
|
-
const tool = new SpeechToTextTool({ openaiApiKey: 'sk-test' });
|
|
367
|
-
mockFetch.mockResolvedValueOnce({
|
|
368
|
-
ok: false,
|
|
369
|
-
status: 404,
|
|
370
|
-
headers: { get: vi.fn().mockReturnValue(null) },
|
|
371
|
-
});
|
|
372
|
-
|
|
373
|
-
const result = await tool.execute({ audioUrl: 'https://example.com/missing.wav' }, ctx);
|
|
374
|
-
|
|
375
|
-
expect(result.success).toBe(false);
|
|
376
|
-
expect(result.error).toContain('Audio download failed (404)');
|
|
377
|
-
});
|
|
378
|
-
|
|
379
|
-
it('respects STT_PROVIDER from the environment', async () => {
|
|
380
|
-
process.env.STT_PROVIDER = 'deepgram';
|
|
381
|
-
process.env.DEEPGRAM_API_KEY = 'dg-env';
|
|
382
|
-
const tool = new SpeechToTextTool({});
|
|
383
|
-
mockFetch.mockResolvedValueOnce({
|
|
384
|
-
ok: true,
|
|
385
|
-
json: async () => ({
|
|
386
|
-
metadata: { duration: 0.8 },
|
|
387
|
-
results: {
|
|
388
|
-
channels: [{ alternatives: [{ transcript: 'env transcript', confidence: 0.91 }] }],
|
|
389
|
-
},
|
|
390
|
-
}),
|
|
391
|
-
});
|
|
392
|
-
|
|
393
|
-
const result = await tool.execute({ audioBase64: Buffer.from('fake-audio').toString('base64') }, ctx);
|
|
394
|
-
|
|
395
|
-
expect(result.success).toBe(true);
|
|
396
|
-
expect(result.output!.provider).toBe('deepgram');
|
|
397
|
-
});
|
|
398
|
-
});
|