@framers/agentos-ext-voice-synthesis 1.0.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +23 -0
- package/dist/index.d.ts +13 -8
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +33 -7
- package/dist/index.js.map +1 -1
- package/dist/tools/textToSpeech.d.ts +29 -6
- package/dist/tools/textToSpeech.d.ts.map +1 -1
- package/dist/tools/textToSpeech.js +179 -29
- package/dist/tools/textToSpeech.js.map +1 -1
- package/manifest.json +1 -1
- package/package.json +49 -18
- package/src/index.ts +40 -9
- package/src/tools/textToSpeech.ts +215 -30
- package/test/textToSpeech.spec.ts +123 -31
package/LICENSE
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Framers
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
23
|
+
|
package/dist/index.d.ts
CHANGED
|
@@ -1,9 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Voice Synthesis Extension Pack —
|
|
2
|
+
* Voice Synthesis Extension Pack — multi-provider TTS for agents.
|
|
3
|
+
*
|
|
4
|
+
* Supports: OpenAI TTS, ElevenLabs, Ollama (local).
|
|
5
|
+
* Auto-detects available provider from API keys.
|
|
3
6
|
*/
|
|
4
|
-
import { TextToSpeechTool } from './tools/textToSpeech.js';
|
|
7
|
+
import { TextToSpeechTool, type TTSProvider } from './tools/textToSpeech.js';
|
|
5
8
|
export interface VoiceSynthesisExtensionOptions {
|
|
6
9
|
elevenLabsApiKey?: string;
|
|
10
|
+
openaiApiKey?: string;
|
|
11
|
+
openaiBaseUrl?: string;
|
|
12
|
+
ollamaBaseUrl?: string;
|
|
13
|
+
defaultProvider?: TTSProvider;
|
|
7
14
|
priority?: number;
|
|
8
15
|
}
|
|
9
16
|
export declare function createExtensionPack(context: any): {
|
|
@@ -14,14 +21,12 @@ export declare function createExtensionPack(context: any): {
|
|
|
14
21
|
kind: "tool";
|
|
15
22
|
priority: number;
|
|
16
23
|
payload: TextToSpeechTool;
|
|
17
|
-
requiredSecrets:
|
|
18
|
-
id: string;
|
|
19
|
-
}[];
|
|
24
|
+
requiredSecrets: never[];
|
|
20
25
|
}[];
|
|
21
|
-
onActivate: () => Promise<
|
|
22
|
-
onDeactivate: () => Promise<
|
|
26
|
+
onActivate: () => Promise<void>;
|
|
27
|
+
onDeactivate: () => Promise<void>;
|
|
23
28
|
};
|
|
24
29
|
export { TextToSpeechTool };
|
|
25
|
-
export type { TTSInput, TTSOutput } from './tools/textToSpeech.js';
|
|
30
|
+
export type { TTSInput, TTSOutput, TTSConfig, TTSProvider } from './tools/textToSpeech.js';
|
|
26
31
|
export default createExtensionPack;
|
|
27
32
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,EAAE,gBAAgB,EAAkB,KAAK,WAAW,EAAE,MAAM,yBAAyB,CAAC;AAE7F,MAAM,WAAW,8BAA8B;IAC7C,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,eAAe,CAAC,EAAE,WAAW,CAAC;IAC9B,QAAQ,CAAC,EAAE,MAAM,CAAC;CACnB;AAED,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,GAAG;;;;;;;;;;;;EAsC/C;AAED,OAAO,EAAE,gBAAgB,EAAE,CAAC;AAC5B,YAAY,EAAE,QAAQ,EAAE,SAAS,EAAE,SAAS,EAAE,WAAW,EAAE,MAAM,yBAAyB,CAAC;AAC3F,eAAe,mBAAmB,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -1,19 +1,45 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Voice Synthesis Extension Pack —
|
|
2
|
+
* Voice Synthesis Extension Pack — multi-provider TTS for agents.
|
|
3
|
+
*
|
|
4
|
+
* Supports: OpenAI TTS, ElevenLabs, Ollama (local).
|
|
5
|
+
* Auto-detects available provider from API keys.
|
|
3
6
|
*/
|
|
4
7
|
import { TextToSpeechTool } from './tools/textToSpeech.js';
|
|
5
8
|
export function createExtensionPack(context) {
|
|
6
9
|
const options = (context.options || {});
|
|
7
|
-
const
|
|
8
|
-
|
|
10
|
+
const config = {
|
|
11
|
+
openaiApiKey: options.openaiApiKey || context.getSecret?.('openai.apiKey') || process.env.OPENAI_API_KEY,
|
|
12
|
+
openaiBaseUrl: options.openaiBaseUrl || process.env.OPENAI_BASE_URL,
|
|
13
|
+
elevenLabsApiKey: options.elevenLabsApiKey || context.getSecret?.('elevenlabs.apiKey') || process.env.ELEVENLABS_API_KEY,
|
|
14
|
+
ollamaBaseUrl: options.ollamaBaseUrl || process.env.OLLAMA_BASE_URL,
|
|
15
|
+
defaultProvider: options.defaultProvider || process.env.TTS_PROVIDER || 'auto',
|
|
16
|
+
};
|
|
17
|
+
const tool = new TextToSpeechTool(config);
|
|
18
|
+
// Determine which providers are available for the activation message
|
|
19
|
+
const providers = [];
|
|
20
|
+
if (config.openaiApiKey)
|
|
21
|
+
providers.push('OpenAI');
|
|
22
|
+
if (config.elevenLabsApiKey)
|
|
23
|
+
providers.push('ElevenLabs');
|
|
24
|
+
providers.push('Ollama (local fallback)');
|
|
9
25
|
return {
|
|
10
26
|
name: '@framers/agentos-ext-voice-synthesis',
|
|
11
|
-
version: '
|
|
27
|
+
version: '2.0.0',
|
|
12
28
|
descriptors: [
|
|
13
|
-
{
|
|
29
|
+
{
|
|
30
|
+
id: tool.name,
|
|
31
|
+
kind: 'tool',
|
|
32
|
+
priority: options.priority || 50,
|
|
33
|
+
payload: tool,
|
|
34
|
+
requiredSecrets: [],
|
|
35
|
+
},
|
|
14
36
|
],
|
|
15
|
-
onActivate: async () =>
|
|
16
|
-
|
|
37
|
+
onActivate: async () => {
|
|
38
|
+
context.logger?.info?.(`Voice Synthesis activated — providers: ${providers.join(', ')}`);
|
|
39
|
+
},
|
|
40
|
+
onDeactivate: async () => {
|
|
41
|
+
context.logger?.info?.('Voice Synthesis deactivated');
|
|
42
|
+
},
|
|
17
43
|
};
|
|
18
44
|
}
|
|
19
45
|
export { TextToSpeechTool };
|
package/dist/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,EAAE,gBAAgB,EAAoC,MAAM,yBAAyB,CAAC;AAW7F,MAAM,UAAU,mBAAmB,CAAC,OAAY;IAC9C,MAAM,OAAO,GAAG,CAAC,OAAO,CAAC,OAAO,IAAI,EAAE,CAAmC,CAAC;IAE1E,MAAM,MAAM,GAAc;QACxB,YAAY,EAAE,OAAO,CAAC,YAAY,IAAI,OAAO,CAAC,SAAS,EAAE,CAAC,eAAe,CAAC,IAAI,OAAO,CAAC,GAAG,CAAC,cAAc;QACxG,aAAa,EAAE,OAAO,CAAC,aAAa,IAAI,OAAO,CAAC,GAAG,CAAC,eAAe;QACnE,gBAAgB,EAAE,OAAO,CAAC,gBAAgB,IAAI,OAAO,CAAC,SAAS,EAAE,CAAC,mBAAmB,CAAC,IAAI,OAAO,CAAC,GAAG,CAAC,kBAAkB;QACxH,aAAa,EAAE,OAAO,CAAC,aAAa,IAAI,OAAO,CAAC,GAAG,CAAC,eAAe;QACnE,eAAe,EAAE,OAAO,CAAC,eAAe,IAAK,OAAO,CAAC,GAAG,CAAC,YAA4B,IAAI,MAAM;KAChG,CAAC;IAEF,MAAM,IAAI,GAAG,IAAI,gBAAgB,CAAC,MAAM,CAAC,CAAC;IAE1C,qEAAqE;IACrE,MAAM,SAAS,GAAa,EAAE,CAAC;IAC/B,IAAI,MAAM,CAAC,YAAY;QAAE,SAAS,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC;IAClD,IAAI,MAAM,CAAC,gBAAgB;QAAE,SAAS,CAAC,IAAI,CAAC,YAAY,CAAC,CAAC;IAC1D,SAAS,CAAC,IAAI,CAAC,yBAAyB,CAAC,CAAC;IAE1C,OAAO;QACL,IAAI,EAAE,sCAAsC;QAC5C,OAAO,EAAE,OAAO;QAChB,WAAW,EAAE;YACX;gBACE,EAAE,EAAE,IAAI,CAAC,IAAI;gBACb,IAAI,EAAE,MAAe;gBACrB,QAAQ,EAAE,OAAO,CAAC,QAAQ,IAAI,EAAE;gBAChC,OAAO,EAAE,IAAI;gBACb,eAAe,EAAE,EAAE;aACpB;SACF;QACD,UAAU,EAAE,KAAK,IAAI,EAAE;YACrB,OAAO,CAAC,MAAM,EAAE,IAAI,EAAE,CAAC,0CAA0C,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QAC3F,CAAC;QACD,YAAY,EAAE,KAAK,IAAI,EAAE;YACvB,OAAO,CAAC,MAAM,EAAE,IAAI,EAAE,CAAC,6BAA6B,CAAC,CAAC;QACxD,CAAC;KACF,CAAC;AACJ,CAAC;AAED,OAAO,EAAE,gBAAgB,EAAE,CAAC;AAE5B,eAAe,mBAAmB,CAAC"}
|
|
@@ -1,34 +1,57 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* Multi-provider TTS Tool — text-to-speech synthesis.
|
|
3
|
+
*
|
|
4
|
+
* Supports: OpenAI TTS, ElevenLabs, Ollama (local), any OpenAI-compatible TTS API.
|
|
5
|
+
* Auto-detects available provider from API keys in environment.
|
|
3
6
|
*/
|
|
4
|
-
import type { ITool, ToolExecutionContext, ToolExecutionResult, JSONSchemaObject } from '
|
|
7
|
+
import type { ITool, ToolExecutionContext, ToolExecutionResult, JSONSchemaObject } from '@framers/agentos';
|
|
8
|
+
export type TTSProvider = 'openai' | 'elevenlabs' | 'ollama' | 'auto';
|
|
5
9
|
export interface TTSInput {
|
|
6
10
|
text: string;
|
|
7
11
|
voice?: string;
|
|
8
12
|
model?: string;
|
|
13
|
+
provider?: TTSProvider;
|
|
14
|
+
/** ElevenLabs-specific */
|
|
9
15
|
stability?: number;
|
|
16
|
+
/** ElevenLabs-specific */
|
|
10
17
|
similarity_boost?: number;
|
|
18
|
+
/** OpenAI-specific: speed 0.25-4.0 */
|
|
19
|
+
speed?: number;
|
|
20
|
+
/** Output format: mp3, opus, aac, flac, wav */
|
|
21
|
+
format?: string;
|
|
11
22
|
}
|
|
12
23
|
export interface TTSOutput {
|
|
13
24
|
text: string;
|
|
14
25
|
voice: string;
|
|
15
26
|
model: string;
|
|
27
|
+
provider: string;
|
|
16
28
|
audioBase64: string;
|
|
17
29
|
contentType: string;
|
|
18
30
|
durationEstimateMs: number;
|
|
19
31
|
}
|
|
32
|
+
export interface TTSConfig {
|
|
33
|
+
openaiApiKey?: string;
|
|
34
|
+
openaiBaseUrl?: string;
|
|
35
|
+
elevenLabsApiKey?: string;
|
|
36
|
+
ollamaBaseUrl?: string;
|
|
37
|
+
defaultProvider?: TTSProvider;
|
|
38
|
+
}
|
|
20
39
|
export declare class TextToSpeechTool implements ITool<TTSInput, TTSOutput> {
|
|
21
|
-
readonly id = "
|
|
40
|
+
readonly id = "tts-multi-provider-v1";
|
|
22
41
|
readonly name = "text_to_speech";
|
|
23
42
|
readonly displayName = "Text to Speech";
|
|
24
43
|
readonly description: string;
|
|
25
44
|
readonly category = "media";
|
|
26
|
-
readonly version = "
|
|
45
|
+
readonly version = "2.0.0";
|
|
27
46
|
readonly hasSideEffects = false;
|
|
28
47
|
readonly inputSchema: JSONSchemaObject;
|
|
29
48
|
readonly requiredCapabilities: string[];
|
|
30
|
-
private
|
|
31
|
-
constructor(
|
|
49
|
+
private config;
|
|
50
|
+
constructor(config?: TTSConfig);
|
|
51
|
+
private resolveProvider;
|
|
32
52
|
execute(args: TTSInput, _context: ToolExecutionContext): Promise<ToolExecutionResult<TTSOutput>>;
|
|
53
|
+
private synthesizeOpenAI;
|
|
54
|
+
private synthesizeElevenLabs;
|
|
55
|
+
private synthesizeOllama;
|
|
33
56
|
}
|
|
34
57
|
//# sourceMappingURL=textToSpeech.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"textToSpeech.d.ts","sourceRoot":"","sources":["../../src/tools/textToSpeech.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"textToSpeech.d.ts","sourceRoot":"","sources":["../../src/tools/textToSpeech.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,KAAK,EAAE,KAAK,EAAE,oBAAoB,EAAE,mBAAmB,EAAE,gBAAgB,EAAE,MAAM,kBAAkB,CAAC;AAE3G,MAAM,MAAM,WAAW,GAAG,QAAQ,GAAG,YAAY,GAAG,QAAQ,GAAG,MAAM,CAAC;AAEtE,MAAM,WAAW,QAAQ;IACvB,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,QAAQ,CAAC,EAAE,WAAW,CAAC;IACvB,0BAA0B;IAC1B,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,0BAA0B;IAC1B,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,sCAAsC;IACtC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,+CAA+C;IAC/C,MAAM,CAAC,EAAE,MAAM,CAAC;CACjB;AAED,MAAM,WAAW,SAAS;IACxB,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;IACd,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,CAAC;IACjB,WAAW,EAAE,MAAM,CAAC;IACpB,WAAW,EAAE,MAAM,CAAC;IACpB,kBAAkB,EAAE,MAAM,CAAC;CAC5B;AAiBD,MAAM,WAAW,SAAS;IACxB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,eAAe,CAAC,EAAE,WAAW,CAAC;CAC/B;AAED,qBAAa,gBAAiB,YAAW,KAAK,CAAC,QAAQ,EAAE,SAAS,CAAC;IACjE,QAAQ,CAAC,EAAE,2BAA2B;IACtC,QAAQ,CAAC,IAAI,oBAAoB;IACjC,QAAQ,CAAC,WAAW,oBAAoB;IACxC,QAAQ,CAAC,WAAW,SAG6D;IACjF,QAAQ,CAAC,QAAQ,WAAW;IAC5B,QAAQ,CAAC,OAAO,WAAW;IAC3B,QAAQ,CAAC,cAAc,SAAS;IAEhC,QAAQ,CAAC,WAAW,EAAE,gBAAgB,CA0BpC;IAEF,QAAQ,CAAC,oBAAoB,WAAsB;IAEnD,OAAO,CAAC,MAAM,CAAY;gBAEd,MAAM,CAAC,EAAE,SAAS;IAU9B,OAAO,CAAC,eAAe;IAgBjB,OAAO,CAAC,IAAI,EAAE,QAAQ,EAAE,QAAQ,EAAE,oBAAoB,GAAG,OAAO,CAAC,mBAAmB,CAAC,SAAS,CAAC,CAAC;YA+BxF,gBAAgB;YAuChB,oBAAoB;YA+CpB,gBAAgB;CAwC/B"}
|
|
@@ -1,7 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* Multi-provider TTS Tool — text-to-speech synthesis.
|
|
3
|
+
*
|
|
4
|
+
* Supports: OpenAI TTS, ElevenLabs, Ollama (local), any OpenAI-compatible TTS API.
|
|
5
|
+
* Auto-detects available provider from API keys in environment.
|
|
3
6
|
*/
|
|
4
|
-
|
|
7
|
+
// ── ElevenLabs voice name → ID mapping ──
|
|
8
|
+
const ELEVENLABS_VOICES = {
|
|
5
9
|
rachel: '21m00Tcm4TlvDq8ikWAM',
|
|
6
10
|
domi: 'AZnzlk1XvdvUeBnXmlld',
|
|
7
11
|
bella: 'EXAVITQu4vr4xnSDxMaL',
|
|
@@ -11,62 +15,208 @@ const VOICES = {
|
|
|
11
15
|
adam: 'pNInz6obpgDQGcFmaJgB',
|
|
12
16
|
sam: 'yoZ06aMxZJJ28mfd3POQ',
|
|
13
17
|
};
|
|
18
|
+
// ── OpenAI voice options ──
|
|
19
|
+
const OPENAI_VOICES = ['alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer'];
|
|
14
20
|
export class TextToSpeechTool {
|
|
15
|
-
id = '
|
|
21
|
+
id = 'tts-multi-provider-v1';
|
|
16
22
|
name = 'text_to_speech';
|
|
17
23
|
displayName = 'Text to Speech';
|
|
18
|
-
description = 'Convert text to speech
|
|
19
|
-
'
|
|
24
|
+
description = 'Convert text to speech audio. Supports multiple providers: OpenAI TTS (alloy/echo/fable/onyx/nova/shimmer), ' +
|
|
25
|
+
'ElevenLabs (rachel/domi/bella/antoni/josh/arnold/adam/sam), or local Ollama TTS. ' +
|
|
26
|
+
'Auto-detects available provider from API keys. Returns base64-encoded audio.';
|
|
20
27
|
category = 'media';
|
|
21
|
-
version = '
|
|
28
|
+
version = '2.0.0';
|
|
22
29
|
hasSideEffects = false;
|
|
23
30
|
inputSchema = {
|
|
24
31
|
type: 'object',
|
|
25
32
|
properties: {
|
|
26
|
-
text: { type: 'string', description: 'Text to convert. Max 5000 chars.' },
|
|
27
|
-
voice: {
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
33
|
+
text: { type: 'string', description: 'Text to convert to speech. Max 5000 chars.' },
|
|
34
|
+
voice: {
|
|
35
|
+
type: 'string',
|
|
36
|
+
description: 'Voice name. OpenAI: alloy, echo, fable, onyx, nova (default), shimmer. ' +
|
|
37
|
+
'ElevenLabs: rachel (default), domi, bella, antoni, josh, arnold, adam, sam. ' +
|
|
38
|
+
'Or a custom voice ID.',
|
|
39
|
+
},
|
|
40
|
+
model: {
|
|
41
|
+
type: 'string',
|
|
42
|
+
description: 'TTS model. OpenAI: tts-1 (default), tts-1-hd. ElevenLabs: eleven_monolingual_v1 (default), eleven_multilingual_v2.',
|
|
43
|
+
},
|
|
44
|
+
provider: {
|
|
45
|
+
type: 'string',
|
|
46
|
+
enum: ['openai', 'elevenlabs', 'ollama', 'auto'],
|
|
47
|
+
description: 'TTS provider. Default: auto (detects from available API keys).',
|
|
48
|
+
},
|
|
49
|
+
speed: { type: 'number', minimum: 0.25, maximum: 4.0, description: 'OpenAI speed multiplier (0.25-4.0).' },
|
|
50
|
+
stability: { type: 'number', minimum: 0, maximum: 1, description: 'ElevenLabs voice stability (0-1).' },
|
|
51
|
+
similarity_boost: { type: 'number', minimum: 0, maximum: 1, description: 'ElevenLabs similarity boost (0-1).' },
|
|
52
|
+
format: { type: 'string', enum: ['mp3', 'opus', 'aac', 'flac', 'wav'], description: 'Output audio format.' },
|
|
31
53
|
},
|
|
32
54
|
required: ['text'],
|
|
33
55
|
};
|
|
34
56
|
requiredCapabilities = ['capability:tts'];
|
|
35
|
-
|
|
36
|
-
constructor(
|
|
37
|
-
this.
|
|
57
|
+
config;
|
|
58
|
+
constructor(config) {
|
|
59
|
+
this.config = {
|
|
60
|
+
openaiApiKey: config?.openaiApiKey || process.env.OPENAI_API_KEY || '',
|
|
61
|
+
openaiBaseUrl: config?.openaiBaseUrl || process.env.OPENAI_BASE_URL || 'https://api.openai.com/v1',
|
|
62
|
+
elevenLabsApiKey: config?.elevenLabsApiKey || process.env.ELEVENLABS_API_KEY || '',
|
|
63
|
+
ollamaBaseUrl: config?.ollamaBaseUrl || process.env.OLLAMA_BASE_URL || 'http://localhost:11434',
|
|
64
|
+
defaultProvider: config?.defaultProvider || process.env.TTS_PROVIDER || 'auto',
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
resolveProvider(requested) {
|
|
68
|
+
const pref = requested || this.config.defaultProvider || 'auto';
|
|
69
|
+
if (pref !== 'auto') {
|
|
70
|
+
// Verify the requested provider has credentials
|
|
71
|
+
if (pref === 'openai' && this.config.openaiApiKey)
|
|
72
|
+
return 'openai';
|
|
73
|
+
if (pref === 'elevenlabs' && this.config.elevenLabsApiKey)
|
|
74
|
+
return 'elevenlabs';
|
|
75
|
+
if (pref === 'ollama')
|
|
76
|
+
return 'ollama';
|
|
77
|
+
// Fall through to auto if requested provider isn't configured
|
|
78
|
+
}
|
|
79
|
+
// Auto-detect: prefer OpenAI (cheaper, faster), then ElevenLabs, then Ollama
|
|
80
|
+
if (this.config.openaiApiKey)
|
|
81
|
+
return 'openai';
|
|
82
|
+
if (this.config.elevenLabsApiKey)
|
|
83
|
+
return 'elevenlabs';
|
|
84
|
+
return 'ollama'; // Local fallback — may or may not have TTS model
|
|
38
85
|
}
|
|
39
86
|
async execute(args, _context) {
|
|
40
|
-
if (!this.apiKey)
|
|
41
|
-
return { success: false, error: 'ELEVENLABS_API_KEY not configured.' };
|
|
42
87
|
const text = args.text.slice(0, 5000);
|
|
43
|
-
const
|
|
88
|
+
const provider = this.resolveProvider(args.provider);
|
|
89
|
+
if (!provider) {
|
|
90
|
+
return {
|
|
91
|
+
success: false,
|
|
92
|
+
error: 'No TTS provider available. Set one of: OPENAI_API_KEY, ELEVENLABS_API_KEY, or configure Ollama with a TTS model. ' +
|
|
93
|
+
'Get an OpenAI key at https://platform.openai.com/api-keys or ElevenLabs at https://elevenlabs.io',
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
try {
|
|
97
|
+
switch (provider) {
|
|
98
|
+
case 'openai':
|
|
99
|
+
return await this.synthesizeOpenAI(text, args);
|
|
100
|
+
case 'elevenlabs':
|
|
101
|
+
return await this.synthesizeElevenLabs(text, args);
|
|
102
|
+
case 'ollama':
|
|
103
|
+
return await this.synthesizeOllama(text, args);
|
|
104
|
+
default:
|
|
105
|
+
return { success: false, error: `Unknown TTS provider: ${provider}` };
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
catch (err) {
|
|
109
|
+
return { success: false, error: `TTS failed (${provider}): ${err.message}` };
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
// ── OpenAI TTS ──
|
|
113
|
+
async synthesizeOpenAI(text, args) {
|
|
114
|
+
const voice = args.voice && OPENAI_VOICES.includes(args.voice) ? args.voice : 'nova';
|
|
115
|
+
const model = args.model || 'tts-1';
|
|
116
|
+
const format = args.format || 'mp3';
|
|
117
|
+
const response = await fetch(`${this.config.openaiBaseUrl}/audio/speech`, {
|
|
118
|
+
method: 'POST',
|
|
119
|
+
headers: {
|
|
120
|
+
Authorization: `Bearer ${this.config.openaiApiKey}`,
|
|
121
|
+
'Content-Type': 'application/json',
|
|
122
|
+
},
|
|
123
|
+
body: JSON.stringify({
|
|
124
|
+
model,
|
|
125
|
+
voice,
|
|
126
|
+
input: text,
|
|
127
|
+
response_format: format,
|
|
128
|
+
speed: args.speed,
|
|
129
|
+
}),
|
|
130
|
+
});
|
|
131
|
+
if (!response.ok) {
|
|
132
|
+
const err = await response.text();
|
|
133
|
+
return { success: false, error: `OpenAI TTS error (${response.status}): ${err.slice(0, 300)}` };
|
|
134
|
+
}
|
|
135
|
+
const buf = await response.arrayBuffer();
|
|
136
|
+
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
137
|
+
const contentType = format === 'opus' ? 'audio/opus' : format === 'wav' ? 'audio/wav' : 'audio/mpeg';
|
|
138
|
+
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
139
|
+
return {
|
|
140
|
+
success: true,
|
|
141
|
+
output: { text, voice, model, provider: 'openai', audioBase64, contentType, durationEstimateMs },
|
|
142
|
+
contentType,
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
// ── ElevenLabs TTS ──
|
|
146
|
+
async synthesizeElevenLabs(text, args) {
|
|
147
|
+
const voiceId = ELEVENLABS_VOICES[(args.voice || 'rachel').toLowerCase()] || args.voice || ELEVENLABS_VOICES.rachel;
|
|
44
148
|
const model = args.model || 'eleven_monolingual_v1';
|
|
149
|
+
const response = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`, {
|
|
150
|
+
method: 'POST',
|
|
151
|
+
headers: {
|
|
152
|
+
'xi-api-key': this.config.elevenLabsApiKey,
|
|
153
|
+
'Content-Type': 'application/json',
|
|
154
|
+
Accept: 'audio/mpeg',
|
|
155
|
+
},
|
|
156
|
+
body: JSON.stringify({
|
|
157
|
+
text,
|
|
158
|
+
model_id: model,
|
|
159
|
+
voice_settings: {
|
|
160
|
+
stability: args.stability ?? 0.5,
|
|
161
|
+
similarity_boost: args.similarity_boost ?? 0.75,
|
|
162
|
+
},
|
|
163
|
+
}),
|
|
164
|
+
});
|
|
165
|
+
if (!response.ok) {
|
|
166
|
+
const err = await response.text();
|
|
167
|
+
return { success: false, error: `ElevenLabs error (${response.status}): ${err.slice(0, 300)}` };
|
|
168
|
+
}
|
|
169
|
+
const buf = await response.arrayBuffer();
|
|
170
|
+
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
171
|
+
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
172
|
+
return {
|
|
173
|
+
success: true,
|
|
174
|
+
output: {
|
|
175
|
+
text,
|
|
176
|
+
voice: args.voice || 'rachel',
|
|
177
|
+
model,
|
|
178
|
+
provider: 'elevenlabs',
|
|
179
|
+
audioBase64,
|
|
180
|
+
contentType: 'audio/mpeg',
|
|
181
|
+
durationEstimateMs,
|
|
182
|
+
},
|
|
183
|
+
contentType: 'audio/mpeg',
|
|
184
|
+
};
|
|
185
|
+
}
|
|
186
|
+
// ── Ollama TTS (local, experimental) ──
|
|
187
|
+
async synthesizeOllama(text, args) {
|
|
188
|
+
// Ollama doesn't natively support TTS yet, but some models (e.g., bark, piper)
|
|
189
|
+
// can be served via OpenAI-compatible endpoints. Try the OpenAI-compat path.
|
|
190
|
+
const voice = args.voice || 'default';
|
|
191
|
+
const model = args.model || 'tts'; // User must have a TTS model loaded
|
|
45
192
|
try {
|
|
46
|
-
const response = await fetch(
|
|
193
|
+
const response = await fetch(`${this.config.ollamaBaseUrl}/v1/audio/speech`, {
|
|
47
194
|
method: 'POST',
|
|
48
|
-
headers: { '
|
|
49
|
-
body: JSON.stringify({
|
|
50
|
-
text,
|
|
51
|
-
model_id: model,
|
|
52
|
-
voice_settings: { stability: args.stability ?? 0.5, similarity_boost: args.similarity_boost ?? 0.75 },
|
|
53
|
-
}),
|
|
195
|
+
headers: { 'Content-Type': 'application/json' },
|
|
196
|
+
body: JSON.stringify({ model, voice, input: text }),
|
|
54
197
|
});
|
|
55
198
|
if (!response.ok) {
|
|
56
|
-
|
|
57
|
-
|
|
199
|
+
return {
|
|
200
|
+
success: false,
|
|
201
|
+
error: `Ollama TTS not available (${response.status}). Ollama doesn't natively support TTS yet. ` +
|
|
202
|
+
'Set OPENAI_API_KEY or ELEVENLABS_API_KEY for cloud TTS, or use a dedicated local TTS server.',
|
|
203
|
+
};
|
|
58
204
|
}
|
|
59
205
|
const buf = await response.arrayBuffer();
|
|
60
206
|
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
61
207
|
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
62
208
|
return {
|
|
63
209
|
success: true,
|
|
64
|
-
output: { text, voice
|
|
210
|
+
output: { text, voice, model, provider: 'ollama', audioBase64, contentType: 'audio/mpeg', durationEstimateMs },
|
|
65
211
|
contentType: 'audio/mpeg',
|
|
66
212
|
};
|
|
67
213
|
}
|
|
68
|
-
catch
|
|
69
|
-
return {
|
|
214
|
+
catch {
|
|
215
|
+
return {
|
|
216
|
+
success: false,
|
|
217
|
+
error: 'Ollama TTS endpoint not reachable. Ollama doesn\'t natively support TTS yet. ' +
|
|
218
|
+
'Set OPENAI_API_KEY or ELEVENLABS_API_KEY for cloud TTS.',
|
|
219
|
+
};
|
|
70
220
|
}
|
|
71
221
|
}
|
|
72
222
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"textToSpeech.js","sourceRoot":"","sources":["../../src/tools/textToSpeech.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"textToSpeech.js","sourceRoot":"","sources":["../../src/tools/textToSpeech.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AA+BH,2CAA2C;AAC3C,MAAM,iBAAiB,GAA2B;IAChD,MAAM,EAAE,sBAAsB;IAC9B,IAAI,EAAE,sBAAsB;IAC5B,KAAK,EAAE,sBAAsB;IAC7B,MAAM,EAAE,sBAAsB;IAC9B,IAAI,EAAE,sBAAsB;IAC5B,MAAM,EAAE,sBAAsB;IAC9B,IAAI,EAAE,sBAAsB;IAC5B,GAAG,EAAE,sBAAsB;CAC5B,CAAC;AAEF,6BAA6B;AAC7B,MAAM,aAAa,GAAG,CAAC,OAAO,EAAE,MAAM,EAAE,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,SAAS,CAAC,CAAC;AAU5E,MAAM,OAAO,gBAAgB;IAClB,EAAE,GAAG,uBAAuB,CAAC;IAC7B,IAAI,GAAG,gBAAgB,CAAC;IACxB,WAAW,GAAG,gBAAgB,CAAC;IAC/B,WAAW,GAClB,8GAA8G;QAC9G,mFAAmF;QACnF,8EAA8E,CAAC;IACxE,QAAQ,GAAG,OAAO,CAAC;IACnB,OAAO,GAAG,OAAO,CAAC;IAClB,cAAc,GAAG,KAAK,CAAC;IAEvB,WAAW,GAAqB;QACvC,IAAI,EAAE,QAAQ;QACd,UAAU,EAAE;YACV,IAAI,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,WAAW,EAAE,4CAA4C,EAAE;YACnF,KAAK,EAAE;gBACL,IAAI,EAAE,QAAQ;gBACd,WAAW,EACT,yEAAyE;oBACzE,8EAA8E;oBAC9E,uBAAuB;aAC1B;YACD,KAAK,EAAE;gBACL,IAAI,EAAE,QAAQ;gBACd,WAAW,EAAE,oHAAoH;aAClI;YACD,QAAQ,EAAE;gBACR,IAAI,EAAE,QAAQ;gBACd,IAAI,EAAE,CAAC,QAAQ,EAAE,YAAY,EAAE,QAAQ,EAAE,MAAM,CAAC;gBAChD,WAAW,EAAE,gEAAgE;aAC9E;YACD,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,IAAI,EAAE,OAAO,EAAE,GAAG,EAAE,WAAW,EAAE,qCAAqC,EAAE;YAC1G,SAAS,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,WAAW,EAAE,mCAAmC,EAAE;YACvG,gBAAgB,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,WAAW,EAAE,oCAAoC,EAAE;YAC/G,MAAM,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,IAAI,EAAE,CAAC,KAAK,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,KAAK,CAAC,EAAE,WAAW,EAAE,sBAAsB,EAAE;SAC7G;QACD,QAAQ,EAAE,CAAC,MAAM,CAAC;KACnB,CAAC;IAEO,oBAAoB,GAAG,CAAC,gBAAgB,CAAC,CAAC;IAE3C,MAAM,CAAY;IAE1B,YAAY,MAAkB;QAC5B,IAAI,CAAC,MAAM,GAAG;YACZ,YAAY,EAAE,MAAM,EAAE,YAAY,IAAI,OAAO,CAAC,GAAG,CAAC,cAAc,IAAI,EAAE;YACtE,aAAa,EAAE,MAAM,EAAE,aAAa,IAAI,OAAO,CAAC,GAAG,CAAC,eAAe,IAAI,2BAA2B;YAClG,gBAAgB,EAAE,MAAM,EAAE,gBAAgB,IAAI,OAAO,CAAC,GAAG,CAAC,kBAAkB,IAAI,EAAE;YAClF,aAAa,EAAE,MAAM,EAAE,aAAa,IAAI,OAAO,CAAC,GAAG,CAAC,eAAe,IAAI,wBAAwB;YAC/F,eAAe,EAAE,MAAM,EAAE,eAAe,IAAK,OAAO,CAAC,GAAG,CAAC,YAA4B,IAAI,MAAM;SAChG,CAAC;IACJ,CAAC;IAEO,eAAe,CAAC,SAAuB;QAC7C,MAAM,IAAI,GAAG,SAAS,IAAI,IAAI,CAAC,MAAM,CAAC,eAAe,IAAI,MAAM,CAAC;QAChE,IAAI,IAAI,KAAK,MAAM,EAAE,CAAC;YACpB,gDAAgD;YAChD,IAAI,IAAI,KAAK,QAAQ,IAAI,IAAI,CAAC,MAAM,CAAC,YAAY;gBAAE,OAAO,QAAQ,CAAC;YACnE,IAAI,IAAI,KAAK,YAAY,IAAI,IAAI,CAAC,MAAM,CAAC,gBAAgB;gBAAE,OAAO,YAAY,CAAC;YAC/E,IAAI,IAAI,KAAK,QAAQ;gBAAE,OAAO,QAAQ,CAAC;YACvC,8DAA8D;QAChE,CAAC;QAED,6EAA6E;QAC7E,IAAI,IAAI,CAAC,MAAM,CAAC,YAAY;YAAE,OAAO,QAAQ,CAAC;QAC9C,IAAI,IAAI,CAAC,MAAM,CAAC,gBAAgB;YAAE,OAAO,YAAY,CAAC;QACtD,OAAO,QAAQ,CAAC,CAAC,iDAAiD;IACpE,CAAC;IAED,KAAK,CAAC,OAAO,CAAC,IAAc,EAAE,QAA8B;QAC1D,MAAM,IAAI,GAAG,IAAI,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,CAAC;QACtC,MAAM,QAAQ,GAAG,IAAI,CAAC,eAAe,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC;QAErD,IAAI,CAAC,QAAQ,EAAE,CAAC;YACd,OAAO;gBACL,OAAO,EAAE,KAAK;gBACd,KAAK,EACH,mHAAmH;oBACnH,kGAAkG;aACrG,CAAC;QACJ,CAAC;QAED,IAAI,CAAC;YACH,QAAQ,QAAQ,EAAE,CAAC;gBACjB,KAAK,QAAQ;oBACX,OAAO,MAAM,IAAI,CAAC,gBAAgB,CAAC,IAAI,EAAE,IAAI,CAAC,CAAC;gBACjD,KAAK,YAAY;oBACf,OAAO,MAAM,IAAI,CAAC,oBAAoB,CAAC,IAAI,EAAE,IAAI,CAAC,CAAC;gBACrD,KAAK,QAAQ;oBACX,OAAO,MAAM,IAAI,CAAC,gBAAgB,CAAC,IAAI,EAAE,IAAI,CAAC,CAAC;gBACjD;oBACE,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,yBAAyB,QAAQ,EAAE,EAAE,CAAC;YAC1E,CAAC;QACH,CAAC;QAAC,OAAO,GAAQ,EAAE,CAAC;YAClB,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,eAAe,QAAQ,MAAM,GAAG,CAAC,OAAO,EAAE,EAAE,CAAC;QAC/E,CAAC;IACH,CAAC;IAED,mBAAmB;IAEX,KAAK,CAAC,gBAAgB,CAAC,IAAY,EAAE,IAAc;QACzD,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,aAAa,CAAC,QAAQ,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC,MAAM,CAAC;QACrF,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,OAAO,CAAC;QACpC,MAAM,MAAM,GAAG,IAAI,CAAC,MAAM,IAAI,KAAK,CAAC;QAEpC,MAAM,QAAQ,GAAG,MAAM,KAAK,CAAC,GAAG,IAAI,CAAC,MAAM,CAAC,aAAa,eAAe,EAAE;YACxE,MAAM,EAAE,MAAM;YACd,OAAO,EAAE;gBACP,aAAa,EAAE,UAAU,IAAI,CAAC,MAAM,CAAC,YAAY,EAAE;gBACnD,cAAc,EAAE,kBAAkB;aACnC;YACD,IAAI,EAAE,IAAI,CAAC,SAAS,CAAC;gBACnB,KAAK;gBACL,KAAK;gBACL,KAAK,EAAE,IAAI;gBACX,eAAe,EAAE,MAAM;gBACvB,KAAK,EAAE,IAAI,CAAC,KAAK;aAClB,CAAC;SACH,CAAC,CAAC;QAEH,IAAI,CAAC,QAAQ,CAAC,EAAE,EAAE,CAAC;YACjB,MAAM,GAAG,GAAG,MAAM,QAAQ,CAAC,IAAI,EAAE,CAAC;YAClC,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,qBAAqB,QAAQ,CAAC,MAAM,MAAM,GAAG,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,EAAE,EAAE,CAAC;QAClG,CAAC;QAED,MAAM,GAAG,GAAG,MAAM,QAAQ,CAAC,WAAW,EAAE,CAAC;QACzC,MAAM,WAAW,GAAG,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,QAAQ,CAAC,CAAC;QACxD,MAAM,WAAW,GAAG,MAAM,KAAK,MAAM,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,MAAM,KAAK,KAAK,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC,CAAC,YAAY,CAAC;QACrG,MAAM,kBAAkB,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,GAAG,GAAG,CAAC,GAAG,EAAE,GAAG,IAAI,CAAC,CAAC;QAEpF,OAAO;YACL,OAAO,EAAE,IAAI;YACb,MAAM,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,KAAK,EAAE,QAAQ,EAAE,QAAQ,EAAE,WAAW,EAAE,WAAW,EAAE,kBAAkB,EAAE;YAChG,WAAW;SACZ,CAAC;IACJ,CAAC;IAED,uBAAuB;IAEf,KAAK,CAAC,oBAAoB,CAAC,IAAY,EAAE,IAAc;QAC7D,MAAM,OAAO,GAAG,iBAAiB,CAAC,CAAC,IAAI,CAAC,KAAK,IAAI,QAAQ,CAAC,CAAC,WAAW,EAAE,CAAC,IAAI,IAAI,CAAC,KAAK,IAAI,iBAAiB,CAAC,MAAM,CAAC;QACpH,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,uBAAuB,CAAC;QAEpD,MAAM,QAAQ,GAAG,MAAM,KAAK,CAAC,+CAA+C,OAAO,EAAE,EAAE;YACrF,MAAM,EAAE,MAAM;YACd,OAAO,EAAE;gBACP,YAAY,EAAE,IAAI,CAAC,MAAM,CAAC,gBAAiB;gBAC3C,cAAc,EAAE,kBAAkB;gBAClC,MAAM,EAAE,YAAY;aACrB;YACD,IAAI,EAAE,IAAI,CAAC,SAAS,CAAC;gBACnB,IAAI;gBACJ,QAAQ,EAAE,KAAK;gBACf,cAAc,EAAE;oBACd,SAAS,EAAE,IAAI,CAAC,SAAS,IAAI,GAAG;oBAChC,gBAAgB,EAAE,IAAI,CAAC,gBAAgB,IAAI,IAAI;iBAChD;aACF,CAAC;SACH,CAAC,CAAC;QAEH,IAAI,CAAC,QAAQ,CAAC,EAAE,EAAE,CAAC;YACjB,MAAM,GAAG,GAAG,MAAM,QAAQ,CAAC,IAAI,EAAE,CAAC;YAClC,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,qBAAqB,QAAQ,CAAC,MAAM,MAAM,GAAG,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,EAAE,EAAE,CAAC;QAClG,CAAC;QAED,MAAM,GAAG,GAAG,MAAM,QAAQ,CAAC,WAAW,EAAE,CAAC;QACzC,MAAM,WAAW,GAAG,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,QAAQ,CAAC,CAAC;QACxD,MAAM,kBAAkB,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,GAAG,GAAG,CAAC,GAAG,EAAE,GAAG,IAAI,CAAC,CAAC;QAEpF,OAAO;YACL,OAAO,EAAE,IAAI;YACb,MAAM,EAAE;gBACN,IAAI;gBACJ,KAAK,EAAE,IAAI,CAAC,KAAK,IAAI,QAAQ;gBAC7B,KAAK;gBACL,QAAQ,EAAE,YAAY;gBACtB,WAAW;gBACX,WAAW,EAAE,YAAY;gBACzB,kBAAkB;aACnB;YACD,WAAW,EAAE,YAAY;SAC1B,CAAC;IACJ,CAAC;IAED,yCAAyC;IAEjC,KAAK,CAAC,gBAAgB,CAAC,IAAY,EAAE,IAAc;QACzD,+EAA+E;QAC/E,6EAA6E;QAC7E,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,SAAS,CAAC;QACtC,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,KAAK,CAAC,CAAC,oCAAoC;QAEvE,IAAI,CAAC;YACH,MAAM,QAAQ,GAAG,MAAM,KAAK,CAAC,GAAG,IAAI,CAAC,MAAM,CAAC,aAAa,kBAAkB,EAAE;gBAC3E,MAAM,EAAE,MAAM;gBACd,OAAO,EAAE,EAAE,cAAc,EAAE,kBAAkB,EAAE;gBAC/C,IAAI,EAAE,IAAI,CAAC,SAAS,CAAC,EAAE,KAAK,EAAE,KAAK,EAAE,KAAK,EAAE,IAAI,EAAE,CAAC;aACpD,CAAC,CAAC;YAEH,IAAI,CAAC,QAAQ,CAAC,EAAE,EAAE,CAAC;gBACjB,OAAO;oBACL,OAAO,EAAE,KAAK;oBACd,KAAK,EACH,6BAA6B,QAAQ,CAAC,MAAM,8CAA8C;wBAC1F,8FAA8F;iBACjG,CAAC;YACJ,CAAC;YAED,MAAM,GAAG,GAAG,MAAM,QAAQ,CAAC,WAAW,EAAE,CAAC;YACzC,MAAM,WAAW,GAAG,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,QAAQ,CAAC,CAAC;YACxD,MAAM,kBAAkB,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,GAAG,GAAG,CAAC,GAAG,EAAE,GAAG,IAAI,CAAC,CAAC;YAEpF,OAAO;gBACL,OAAO,EAAE,IAAI;gBACb,MAAM,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,KAAK,EAAE,QAAQ,EAAE,QAAQ,EAAE,WAAW,EAAE,WAAW,EAAE,YAAY,EAAE,kBAAkB,EAAE;gBAC9G,WAAW,EAAE,YAAY;aAC1B,CAAC;QACJ,CAAC;QAAC,MAAM,CAAC;YACP,OAAO;gBACL,OAAO,EAAE,KAAK;gBACd,KAAK,EACH,+EAA+E;oBAC/E,yDAAyD;aAC5D,CAAC;QACJ,CAAC;IACH,CAAC;CACF"}
|
package/manifest.json
CHANGED
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
"agentosVersion": "^2.0.0",
|
|
11
11
|
"categories": ["media", "communication"],
|
|
12
12
|
"extensions": [
|
|
13
|
-
{ "kind": "tool", "id": "
|
|
13
|
+
{ "kind": "tool", "id": "text_to_speech", "displayName": "Text to Speech", "entry": "./dist/tools/textToSpeech.js" }
|
|
14
14
|
],
|
|
15
15
|
"configuration": {
|
|
16
16
|
"properties": {
|
package/package.json
CHANGED
|
@@ -1,32 +1,54 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@framers/agentos-ext-voice-synthesis",
|
|
3
|
-
"version": "
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "2.0.0",
|
|
4
|
+
"description": "Multi-provider text-to-speech synthesis tool for AgentOS via OpenAI, ElevenLabs, and local Ollama-compatible runtimes",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"types": "dist/index.d.ts",
|
|
7
7
|
"type": "module",
|
|
8
8
|
"sideEffects": false,
|
|
9
9
|
"exports": {
|
|
10
|
-
".": {
|
|
10
|
+
".": {
|
|
11
|
+
"import": "./dist/index.js",
|
|
12
|
+
"types": "./dist/index.d.ts"
|
|
13
|
+
},
|
|
11
14
|
"./manifest.json": "./manifest.json",
|
|
12
15
|
"./package.json": "./package.json"
|
|
13
16
|
},
|
|
14
|
-
"
|
|
15
|
-
"
|
|
16
|
-
"
|
|
17
|
-
"
|
|
18
|
-
"
|
|
19
|
-
"
|
|
20
|
-
"
|
|
21
|
-
"
|
|
17
|
+
"keywords": [
|
|
18
|
+
"agentos",
|
|
19
|
+
"extension",
|
|
20
|
+
"tts",
|
|
21
|
+
"voice",
|
|
22
|
+
"openai",
|
|
23
|
+
"elevenlabs",
|
|
24
|
+
"ollama",
|
|
25
|
+
"speech",
|
|
26
|
+
"media"
|
|
27
|
+
],
|
|
28
|
+
"author": {
|
|
29
|
+
"name": "Framers AI",
|
|
30
|
+
"email": "team@frame.dev",
|
|
31
|
+
"url": "https://frame.dev"
|
|
22
32
|
},
|
|
23
|
-
"
|
|
24
|
-
|
|
25
|
-
|
|
33
|
+
"contributors": [
|
|
34
|
+
{
|
|
35
|
+
"name": "Johnny Dunn",
|
|
36
|
+
"email": "johnnyfived@protonmail.com",
|
|
37
|
+
"url": "https://github.com/jddunn"
|
|
38
|
+
}
|
|
39
|
+
],
|
|
26
40
|
"license": "MIT",
|
|
27
|
-
"repository": {
|
|
28
|
-
|
|
29
|
-
|
|
41
|
+
"repository": {
|
|
42
|
+
"type": "git",
|
|
43
|
+
"url": "https://github.com/framersai/agentos-extensions.git",
|
|
44
|
+
"directory": "registry/curated/media/voice-synthesis"
|
|
45
|
+
},
|
|
46
|
+
"publishConfig": {
|
|
47
|
+
"access": "public"
|
|
48
|
+
},
|
|
49
|
+
"peerDependencies": {
|
|
50
|
+
"@framers/agentos": "^0.1.0"
|
|
51
|
+
},
|
|
30
52
|
"devDependencies": {
|
|
31
53
|
"@framers/agentos": "^0.1.0",
|
|
32
54
|
"@types/node": "^20.12.12",
|
|
@@ -34,5 +56,14 @@
|
|
|
34
56
|
"rimraf": "^5.0.7",
|
|
35
57
|
"typescript": "^5.4.5",
|
|
36
58
|
"vitest": "^1.6.0"
|
|
59
|
+
},
|
|
60
|
+
"scripts": {
|
|
61
|
+
"build": "tsc",
|
|
62
|
+
"test": "vitest run",
|
|
63
|
+
"test:watch": "vitest",
|
|
64
|
+
"test:coverage": "vitest run --coverage",
|
|
65
|
+
"lint": "eslint src --ext .ts",
|
|
66
|
+
"typecheck": "tsc --noEmit",
|
|
67
|
+
"clean": "rimraf dist"
|
|
37
68
|
}
|
|
38
|
-
}
|
|
69
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -1,30 +1,61 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Voice Synthesis Extension Pack —
|
|
2
|
+
* Voice Synthesis Extension Pack — multi-provider TTS for agents.
|
|
3
|
+
*
|
|
4
|
+
* Supports: OpenAI TTS, ElevenLabs, Ollama (local).
|
|
5
|
+
* Auto-detects available provider from API keys.
|
|
3
6
|
*/
|
|
4
7
|
|
|
5
|
-
import { TextToSpeechTool } from './tools/textToSpeech.js';
|
|
8
|
+
import { TextToSpeechTool, type TTSConfig, type TTSProvider } from './tools/textToSpeech.js';
|
|
6
9
|
|
|
7
10
|
export interface VoiceSynthesisExtensionOptions {
|
|
8
11
|
elevenLabsApiKey?: string;
|
|
12
|
+
openaiApiKey?: string;
|
|
13
|
+
openaiBaseUrl?: string;
|
|
14
|
+
ollamaBaseUrl?: string;
|
|
15
|
+
defaultProvider?: TTSProvider;
|
|
9
16
|
priority?: number;
|
|
10
17
|
}
|
|
11
18
|
|
|
12
19
|
export function createExtensionPack(context: any) {
|
|
13
20
|
const options = (context.options || {}) as VoiceSynthesisExtensionOptions;
|
|
14
|
-
|
|
15
|
-
const
|
|
21
|
+
|
|
22
|
+
const config: TTSConfig = {
|
|
23
|
+
openaiApiKey: options.openaiApiKey || context.getSecret?.('openai.apiKey') || process.env.OPENAI_API_KEY,
|
|
24
|
+
openaiBaseUrl: options.openaiBaseUrl || process.env.OPENAI_BASE_URL,
|
|
25
|
+
elevenLabsApiKey: options.elevenLabsApiKey || context.getSecret?.('elevenlabs.apiKey') || process.env.ELEVENLABS_API_KEY,
|
|
26
|
+
ollamaBaseUrl: options.ollamaBaseUrl || process.env.OLLAMA_BASE_URL,
|
|
27
|
+
defaultProvider: options.defaultProvider || (process.env.TTS_PROVIDER as TTSProvider) || 'auto',
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
const tool = new TextToSpeechTool(config);
|
|
31
|
+
|
|
32
|
+
// Determine which providers are available for the activation message
|
|
33
|
+
const providers: string[] = [];
|
|
34
|
+
if (config.openaiApiKey) providers.push('OpenAI');
|
|
35
|
+
if (config.elevenLabsApiKey) providers.push('ElevenLabs');
|
|
36
|
+
providers.push('Ollama (local fallback)');
|
|
16
37
|
|
|
17
38
|
return {
|
|
18
39
|
name: '@framers/agentos-ext-voice-synthesis',
|
|
19
|
-
version: '
|
|
40
|
+
version: '2.0.0',
|
|
20
41
|
descriptors: [
|
|
21
|
-
{
|
|
42
|
+
{
|
|
43
|
+
id: tool.name,
|
|
44
|
+
kind: 'tool' as const,
|
|
45
|
+
priority: options.priority || 50,
|
|
46
|
+
payload: tool,
|
|
47
|
+
requiredSecrets: [],
|
|
48
|
+
},
|
|
22
49
|
],
|
|
23
|
-
onActivate: async () =>
|
|
24
|
-
|
|
50
|
+
onActivate: async () => {
|
|
51
|
+
context.logger?.info?.(`Voice Synthesis activated — providers: ${providers.join(', ')}`);
|
|
52
|
+
},
|
|
53
|
+
onDeactivate: async () => {
|
|
54
|
+
context.logger?.info?.('Voice Synthesis deactivated');
|
|
55
|
+
},
|
|
25
56
|
};
|
|
26
57
|
}
|
|
27
58
|
|
|
28
59
|
export { TextToSpeechTool };
|
|
29
|
-
export type { TTSInput, TTSOutput } from './tools/textToSpeech.js';
|
|
60
|
+
export type { TTSInput, TTSOutput, TTSConfig, TTSProvider } from './tools/textToSpeech.js';
|
|
30
61
|
export default createExtensionPack;
|
|
@@ -1,27 +1,41 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* Multi-provider TTS Tool — text-to-speech synthesis.
|
|
3
|
+
*
|
|
4
|
+
* Supports: OpenAI TTS, ElevenLabs, Ollama (local), any OpenAI-compatible TTS API.
|
|
5
|
+
* Auto-detects available provider from API keys in environment.
|
|
3
6
|
*/
|
|
4
7
|
|
|
5
|
-
import type { ITool, ToolExecutionContext, ToolExecutionResult, JSONSchemaObject } from '
|
|
8
|
+
import type { ITool, ToolExecutionContext, ToolExecutionResult, JSONSchemaObject } from '@framers/agentos';
|
|
9
|
+
|
|
10
|
+
export type TTSProvider = 'openai' | 'elevenlabs' | 'ollama' | 'auto';
|
|
6
11
|
|
|
7
12
|
export interface TTSInput {
|
|
8
13
|
text: string;
|
|
9
14
|
voice?: string;
|
|
10
15
|
model?: string;
|
|
16
|
+
provider?: TTSProvider;
|
|
17
|
+
/** ElevenLabs-specific */
|
|
11
18
|
stability?: number;
|
|
19
|
+
/** ElevenLabs-specific */
|
|
12
20
|
similarity_boost?: number;
|
|
21
|
+
/** OpenAI-specific: speed 0.25-4.0 */
|
|
22
|
+
speed?: number;
|
|
23
|
+
/** Output format: mp3, opus, aac, flac, wav */
|
|
24
|
+
format?: string;
|
|
13
25
|
}
|
|
14
26
|
|
|
15
27
|
export interface TTSOutput {
|
|
16
28
|
text: string;
|
|
17
29
|
voice: string;
|
|
18
30
|
model: string;
|
|
31
|
+
provider: string;
|
|
19
32
|
audioBase64: string;
|
|
20
33
|
contentType: string;
|
|
21
34
|
durationEstimateMs: number;
|
|
22
35
|
}
|
|
23
36
|
|
|
24
|
-
|
|
37
|
+
// ── ElevenLabs voice name → ID mapping ──
|
|
38
|
+
const ELEVENLABS_VOICES: Record<string, string> = {
|
|
25
39
|
rachel: '21m00Tcm4TlvDq8ikWAM',
|
|
26
40
|
domi: 'AZnzlk1XvdvUeBnXmlld',
|
|
27
41
|
bella: 'EXAVITQu4vr4xnSDxMaL',
|
|
@@ -32,58 +46,224 @@ const VOICES: Record<string, string> = {
|
|
|
32
46
|
sam: 'yoZ06aMxZJJ28mfd3POQ',
|
|
33
47
|
};
|
|
34
48
|
|
|
49
|
+
// ── OpenAI voice options ──
|
|
50
|
+
const OPENAI_VOICES = ['alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer'];
|
|
51
|
+
|
|
52
|
+
export interface TTSConfig {
|
|
53
|
+
openaiApiKey?: string;
|
|
54
|
+
openaiBaseUrl?: string;
|
|
55
|
+
elevenLabsApiKey?: string;
|
|
56
|
+
ollamaBaseUrl?: string;
|
|
57
|
+
defaultProvider?: TTSProvider;
|
|
58
|
+
}
|
|
59
|
+
|
|
35
60
|
export class TextToSpeechTool implements ITool<TTSInput, TTSOutput> {
|
|
36
|
-
readonly id = '
|
|
61
|
+
readonly id = 'tts-multi-provider-v1';
|
|
37
62
|
readonly name = 'text_to_speech';
|
|
38
63
|
readonly displayName = 'Text to Speech';
|
|
39
64
|
readonly description =
|
|
40
|
-
'Convert text to speech
|
|
41
|
-
'
|
|
65
|
+
'Convert text to speech audio. Supports multiple providers: OpenAI TTS (alloy/echo/fable/onyx/nova/shimmer), ' +
|
|
66
|
+
'ElevenLabs (rachel/domi/bella/antoni/josh/arnold/adam/sam), or local Ollama TTS. ' +
|
|
67
|
+
'Auto-detects available provider from API keys. Returns base64-encoded audio.';
|
|
42
68
|
readonly category = 'media';
|
|
43
|
-
readonly version = '
|
|
69
|
+
readonly version = '2.0.0';
|
|
44
70
|
readonly hasSideEffects = false;
|
|
45
71
|
|
|
46
72
|
readonly inputSchema: JSONSchemaObject = {
|
|
47
73
|
type: 'object',
|
|
48
74
|
properties: {
|
|
49
|
-
text: { type: 'string', description: 'Text to convert. Max 5000 chars.' },
|
|
50
|
-
voice: {
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
75
|
+
text: { type: 'string', description: 'Text to convert to speech. Max 5000 chars.' },
|
|
76
|
+
voice: {
|
|
77
|
+
type: 'string',
|
|
78
|
+
description:
|
|
79
|
+
'Voice name. OpenAI: alloy, echo, fable, onyx, nova (default), shimmer. ' +
|
|
80
|
+
'ElevenLabs: rachel (default), domi, bella, antoni, josh, arnold, adam, sam. ' +
|
|
81
|
+
'Or a custom voice ID.',
|
|
82
|
+
},
|
|
83
|
+
model: {
|
|
84
|
+
type: 'string',
|
|
85
|
+
description: 'TTS model. OpenAI: tts-1 (default), tts-1-hd. ElevenLabs: eleven_monolingual_v1 (default), eleven_multilingual_v2.',
|
|
86
|
+
},
|
|
87
|
+
provider: {
|
|
88
|
+
type: 'string',
|
|
89
|
+
enum: ['openai', 'elevenlabs', 'ollama', 'auto'],
|
|
90
|
+
description: 'TTS provider. Default: auto (detects from available API keys).',
|
|
91
|
+
},
|
|
92
|
+
speed: { type: 'number', minimum: 0.25, maximum: 4.0, description: 'OpenAI speed multiplier (0.25-4.0).' },
|
|
93
|
+
stability: { type: 'number', minimum: 0, maximum: 1, description: 'ElevenLabs voice stability (0-1).' },
|
|
94
|
+
similarity_boost: { type: 'number', minimum: 0, maximum: 1, description: 'ElevenLabs similarity boost (0-1).' },
|
|
95
|
+
format: { type: 'string', enum: ['mp3', 'opus', 'aac', 'flac', 'wav'], description: 'Output audio format.' },
|
|
54
96
|
},
|
|
55
97
|
required: ['text'],
|
|
56
98
|
};
|
|
57
99
|
|
|
58
100
|
readonly requiredCapabilities = ['capability:tts'];
|
|
59
101
|
|
|
60
|
-
private
|
|
102
|
+
private config: TTSConfig;
|
|
61
103
|
|
|
62
|
-
constructor(
|
|
63
|
-
this.
|
|
104
|
+
constructor(config?: TTSConfig) {
|
|
105
|
+
this.config = {
|
|
106
|
+
openaiApiKey: config?.openaiApiKey || process.env.OPENAI_API_KEY || '',
|
|
107
|
+
openaiBaseUrl: config?.openaiBaseUrl || process.env.OPENAI_BASE_URL || 'https://api.openai.com/v1',
|
|
108
|
+
elevenLabsApiKey: config?.elevenLabsApiKey || process.env.ELEVENLABS_API_KEY || '',
|
|
109
|
+
ollamaBaseUrl: config?.ollamaBaseUrl || process.env.OLLAMA_BASE_URL || 'http://localhost:11434',
|
|
110
|
+
defaultProvider: config?.defaultProvider || (process.env.TTS_PROVIDER as TTSProvider) || 'auto',
|
|
111
|
+
};
|
|
64
112
|
}
|
|
65
113
|
|
|
66
|
-
|
|
67
|
-
|
|
114
|
+
private resolveProvider(requested?: TTSProvider): TTSProvider | null {
|
|
115
|
+
const pref = requested || this.config.defaultProvider || 'auto';
|
|
116
|
+
if (pref !== 'auto') {
|
|
117
|
+
// Verify the requested provider has credentials
|
|
118
|
+
if (pref === 'openai' && this.config.openaiApiKey) return 'openai';
|
|
119
|
+
if (pref === 'elevenlabs' && this.config.elevenLabsApiKey) return 'elevenlabs';
|
|
120
|
+
if (pref === 'ollama') return 'ollama';
|
|
121
|
+
// Fall through to auto if requested provider isn't configured
|
|
122
|
+
}
|
|
68
123
|
|
|
124
|
+
// Auto-detect: prefer OpenAI (cheaper, faster), then ElevenLabs, then Ollama
|
|
125
|
+
if (this.config.openaiApiKey) return 'openai';
|
|
126
|
+
if (this.config.elevenLabsApiKey) return 'elevenlabs';
|
|
127
|
+
return 'ollama'; // Local fallback — may or may not have TTS model
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
async execute(args: TTSInput, _context: ToolExecutionContext): Promise<ToolExecutionResult<TTSOutput>> {
|
|
69
131
|
const text = args.text.slice(0, 5000);
|
|
70
|
-
const
|
|
132
|
+
const provider = this.resolveProvider(args.provider);
|
|
133
|
+
|
|
134
|
+
if (!provider) {
|
|
135
|
+
return {
|
|
136
|
+
success: false,
|
|
137
|
+
error:
|
|
138
|
+
'No TTS provider available. Set one of: OPENAI_API_KEY, ELEVENLABS_API_KEY, or configure Ollama with a TTS model. ' +
|
|
139
|
+
'Get an OpenAI key at https://platform.openai.com/api-keys or ElevenLabs at https://elevenlabs.io',
|
|
140
|
+
};
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
try {
|
|
144
|
+
switch (provider) {
|
|
145
|
+
case 'openai':
|
|
146
|
+
return await this.synthesizeOpenAI(text, args);
|
|
147
|
+
case 'elevenlabs':
|
|
148
|
+
return await this.synthesizeElevenLabs(text, args);
|
|
149
|
+
case 'ollama':
|
|
150
|
+
return await this.synthesizeOllama(text, args);
|
|
151
|
+
default:
|
|
152
|
+
return { success: false, error: `Unknown TTS provider: ${provider}` };
|
|
153
|
+
}
|
|
154
|
+
} catch (err: any) {
|
|
155
|
+
return { success: false, error: `TTS failed (${provider}): ${err.message}` };
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// ── OpenAI TTS ──
|
|
160
|
+
|
|
161
|
+
private async synthesizeOpenAI(text: string, args: TTSInput): Promise<ToolExecutionResult<TTSOutput>> {
|
|
162
|
+
const voice = args.voice && OPENAI_VOICES.includes(args.voice) ? args.voice : 'nova';
|
|
163
|
+
const model = args.model || 'tts-1';
|
|
164
|
+
const format = args.format || 'mp3';
|
|
165
|
+
|
|
166
|
+
const response = await fetch(`${this.config.openaiBaseUrl}/audio/speech`, {
|
|
167
|
+
method: 'POST',
|
|
168
|
+
headers: {
|
|
169
|
+
Authorization: `Bearer ${this.config.openaiApiKey}`,
|
|
170
|
+
'Content-Type': 'application/json',
|
|
171
|
+
},
|
|
172
|
+
body: JSON.stringify({
|
|
173
|
+
model,
|
|
174
|
+
voice,
|
|
175
|
+
input: text,
|
|
176
|
+
response_format: format,
|
|
177
|
+
speed: args.speed,
|
|
178
|
+
}),
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
if (!response.ok) {
|
|
182
|
+
const err = await response.text();
|
|
183
|
+
return { success: false, error: `OpenAI TTS error (${response.status}): ${err.slice(0, 300)}` };
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
const buf = await response.arrayBuffer();
|
|
187
|
+
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
188
|
+
const contentType = format === 'opus' ? 'audio/opus' : format === 'wav' ? 'audio/wav' : 'audio/mpeg';
|
|
189
|
+
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
190
|
+
|
|
191
|
+
return {
|
|
192
|
+
success: true,
|
|
193
|
+
output: { text, voice, model, provider: 'openai', audioBase64, contentType, durationEstimateMs },
|
|
194
|
+
contentType,
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// ── ElevenLabs TTS ──
|
|
199
|
+
|
|
200
|
+
private async synthesizeElevenLabs(text: string, args: TTSInput): Promise<ToolExecutionResult<TTSOutput>> {
|
|
201
|
+
const voiceId = ELEVENLABS_VOICES[(args.voice || 'rachel').toLowerCase()] || args.voice || ELEVENLABS_VOICES.rachel;
|
|
71
202
|
const model = args.model || 'eleven_monolingual_v1';
|
|
72
203
|
|
|
204
|
+
const response = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`, {
|
|
205
|
+
method: 'POST',
|
|
206
|
+
headers: {
|
|
207
|
+
'xi-api-key': this.config.elevenLabsApiKey!,
|
|
208
|
+
'Content-Type': 'application/json',
|
|
209
|
+
Accept: 'audio/mpeg',
|
|
210
|
+
},
|
|
211
|
+
body: JSON.stringify({
|
|
212
|
+
text,
|
|
213
|
+
model_id: model,
|
|
214
|
+
voice_settings: {
|
|
215
|
+
stability: args.stability ?? 0.5,
|
|
216
|
+
similarity_boost: args.similarity_boost ?? 0.75,
|
|
217
|
+
},
|
|
218
|
+
}),
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
if (!response.ok) {
|
|
222
|
+
const err = await response.text();
|
|
223
|
+
return { success: false, error: `ElevenLabs error (${response.status}): ${err.slice(0, 300)}` };
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
const buf = await response.arrayBuffer();
|
|
227
|
+
const audioBase64 = Buffer.from(buf).toString('base64');
|
|
228
|
+
const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
|
|
229
|
+
|
|
230
|
+
return {
|
|
231
|
+
success: true,
|
|
232
|
+
output: {
|
|
233
|
+
text,
|
|
234
|
+
voice: args.voice || 'rachel',
|
|
235
|
+
model,
|
|
236
|
+
provider: 'elevenlabs',
|
|
237
|
+
audioBase64,
|
|
238
|
+
contentType: 'audio/mpeg',
|
|
239
|
+
durationEstimateMs,
|
|
240
|
+
},
|
|
241
|
+
contentType: 'audio/mpeg',
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
// ── Ollama TTS (local, experimental) ──
|
|
246
|
+
|
|
247
|
+
private async synthesizeOllama(text: string, args: TTSInput): Promise<ToolExecutionResult<TTSOutput>> {
|
|
248
|
+
// Ollama doesn't natively support TTS yet, but some models (e.g., bark, piper)
|
|
249
|
+
// can be served via OpenAI-compatible endpoints. Try the OpenAI-compat path.
|
|
250
|
+
const voice = args.voice || 'default';
|
|
251
|
+
const model = args.model || 'tts'; // User must have a TTS model loaded
|
|
252
|
+
|
|
73
253
|
try {
|
|
74
|
-
const response = await fetch(
|
|
254
|
+
const response = await fetch(`${this.config.ollamaBaseUrl}/v1/audio/speech`, {
|
|
75
255
|
method: 'POST',
|
|
76
|
-
headers: { '
|
|
77
|
-
body: JSON.stringify({
|
|
78
|
-
text,
|
|
79
|
-
model_id: model,
|
|
80
|
-
voice_settings: { stability: args.stability ?? 0.5, similarity_boost: args.similarity_boost ?? 0.75 },
|
|
81
|
-
}),
|
|
256
|
+
headers: { 'Content-Type': 'application/json' },
|
|
257
|
+
body: JSON.stringify({ model, voice, input: text }),
|
|
82
258
|
});
|
|
83
259
|
|
|
84
260
|
if (!response.ok) {
|
|
85
|
-
|
|
86
|
-
|
|
261
|
+
return {
|
|
262
|
+
success: false,
|
|
263
|
+
error:
|
|
264
|
+
`Ollama TTS not available (${response.status}). Ollama doesn't natively support TTS yet. ` +
|
|
265
|
+
'Set OPENAI_API_KEY or ELEVENLABS_API_KEY for cloud TTS, or use a dedicated local TTS server.',
|
|
266
|
+
};
|
|
87
267
|
}
|
|
88
268
|
|
|
89
269
|
const buf = await response.arrayBuffer();
|
|
@@ -92,11 +272,16 @@ export class TextToSpeechTool implements ITool<TTSInput, TTSOutput> {
|
|
|
92
272
|
|
|
93
273
|
return {
|
|
94
274
|
success: true,
|
|
95
|
-
output: { text, voice
|
|
275
|
+
output: { text, voice, model, provider: 'ollama', audioBase64, contentType: 'audio/mpeg', durationEstimateMs },
|
|
96
276
|
contentType: 'audio/mpeg',
|
|
97
277
|
};
|
|
98
|
-
} catch
|
|
99
|
-
return {
|
|
278
|
+
} catch {
|
|
279
|
+
return {
|
|
280
|
+
success: false,
|
|
281
|
+
error:
|
|
282
|
+
'Ollama TTS endpoint not reachable. Ollama doesn\'t natively support TTS yet. ' +
|
|
283
|
+
'Set OPENAI_API_KEY or ELEVENLABS_API_KEY for cloud TTS.',
|
|
284
|
+
};
|
|
100
285
|
}
|
|
101
286
|
}
|
|
102
287
|
}
|
|
@@ -1,61 +1,138 @@
|
|
|
1
|
-
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
|
1
|
+
import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest';
|
|
2
2
|
|
|
3
3
|
const mockFetch = vi.fn();
|
|
4
4
|
vi.stubGlobal('fetch', mockFetch);
|
|
5
5
|
|
|
6
|
+
// Clear env vars that affect provider detection
|
|
7
|
+
const savedEnv: Record<string, string | undefined> = {};
|
|
8
|
+
const envKeys = ['OPENAI_API_KEY', 'ELEVENLABS_API_KEY', 'OPENAI_BASE_URL', 'OLLAMA_BASE_URL', 'TTS_PROVIDER'];
|
|
9
|
+
|
|
6
10
|
const { TextToSpeechTool } = await import('../src/tools/textToSpeech.js');
|
|
7
11
|
const { createExtensionPack } = await import('../src/index.js');
|
|
8
12
|
|
|
9
13
|
describe('TextToSpeechTool', () => {
|
|
10
|
-
|
|
14
|
+
const ctx = {} as any;
|
|
11
15
|
|
|
12
16
|
beforeEach(() => {
|
|
13
17
|
vi.clearAllMocks();
|
|
14
|
-
|
|
18
|
+
// Save and clear env
|
|
19
|
+
for (const key of envKeys) {
|
|
20
|
+
savedEnv[key] = process.env[key];
|
|
21
|
+
delete process.env[key];
|
|
22
|
+
}
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
afterEach(() => {
|
|
26
|
+
// Restore env
|
|
27
|
+
for (const key of envKeys) {
|
|
28
|
+
if (savedEnv[key] !== undefined) process.env[key] = savedEnv[key];
|
|
29
|
+
else delete process.env[key];
|
|
30
|
+
}
|
|
15
31
|
});
|
|
16
32
|
|
|
17
33
|
describe('metadata', () => {
|
|
18
34
|
it('has correct id and name', () => {
|
|
19
|
-
|
|
35
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test' });
|
|
36
|
+
expect(tool.id).toBe('tts-multi-provider-v1');
|
|
20
37
|
expect(tool.name).toBe('text_to_speech');
|
|
21
38
|
});
|
|
22
39
|
|
|
23
|
-
it('has valid input schema', () => {
|
|
40
|
+
it('has valid input schema with text required', () => {
|
|
41
|
+
const tool = new TextToSpeechTool({});
|
|
24
42
|
expect(tool.inputSchema.type).toBe('object');
|
|
25
43
|
expect(tool.inputSchema.required).toContain('text');
|
|
26
44
|
});
|
|
27
45
|
|
|
28
46
|
it('has no side effects', () => {
|
|
47
|
+
const tool = new TextToSpeechTool({});
|
|
29
48
|
expect(tool.hasSideEffects).toBe(false);
|
|
30
49
|
});
|
|
31
|
-
});
|
|
32
50
|
|
|
33
|
-
|
|
34
|
-
|
|
51
|
+
it('describes multiple providers', () => {
|
|
52
|
+
const tool = new TextToSpeechTool({});
|
|
53
|
+
expect(tool.description).toContain('OpenAI');
|
|
54
|
+
expect(tool.description).toContain('ElevenLabs');
|
|
55
|
+
});
|
|
56
|
+
});
|
|
35
57
|
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
const
|
|
58
|
+
describe('provider resolution', () => {
|
|
59
|
+
it('falls back to Ollama when no API keys set', async () => {
|
|
60
|
+
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: '' });
|
|
61
|
+
mockFetch.mockRejectedValueOnce(new Error('ECONNREFUSED'));
|
|
62
|
+
const result = await tool.execute({ text: 'Hello' }, ctx);
|
|
39
63
|
expect(result.success).toBe(false);
|
|
40
|
-
expect(result.error).toContain('
|
|
64
|
+
expect(result.error).toContain('Ollama');
|
|
41
65
|
});
|
|
42
66
|
|
|
43
|
-
it('
|
|
44
|
-
const
|
|
45
|
-
mockFetch.mockResolvedValueOnce({
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
67
|
+
it('auto-detects OpenAI when key provided', async () => {
|
|
68
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
69
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
70
|
+
const result = await tool.execute({ text: 'Hello' }, ctx);
|
|
71
|
+
expect(result.success).toBe(true);
|
|
72
|
+
expect(result.output!.provider).toBe('openai');
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
it('auto-detects ElevenLabs when only that key set', async () => {
|
|
76
|
+
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
77
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
78
|
+
const result = await tool.execute({ text: 'Hello' }, ctx);
|
|
79
|
+
expect(result.success).toBe(true);
|
|
80
|
+
expect(result.output!.provider).toBe('elevenlabs');
|
|
81
|
+
});
|
|
49
82
|
|
|
83
|
+
it('respects explicit provider override', async () => {
|
|
84
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: 'el-test' });
|
|
85
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
86
|
+
const result = await tool.execute({ text: 'Hello', provider: 'elevenlabs' }, ctx);
|
|
87
|
+
expect(result.success).toBe(true);
|
|
88
|
+
expect(result.output!.provider).toBe('elevenlabs');
|
|
89
|
+
});
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
describe('OpenAI TTS', () => {
|
|
93
|
+
it('synthesizes with default voice (nova)', async () => {
|
|
94
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
95
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
50
96
|
const result = await tool.execute({ text: 'Hello world' }, ctx);
|
|
51
97
|
expect(result.success).toBe(true);
|
|
52
|
-
expect(result.output!.voice).toBe('
|
|
98
|
+
expect(result.output!.voice).toBe('nova');
|
|
99
|
+
expect(result.output!.provider).toBe('openai');
|
|
53
100
|
expect(result.output!.contentType).toBe('audio/mpeg');
|
|
54
101
|
expect(result.output!.audioBase64).toBeTruthy();
|
|
55
|
-
expect(result.output!.durationEstimateMs).toBeGreaterThan(0);
|
|
56
102
|
});
|
|
57
103
|
|
|
58
|
-
it('
|
|
104
|
+
it('sends correct request to OpenAI API', async () => {
|
|
105
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
106
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
107
|
+
await tool.execute({ text: 'Test', voice: 'shimmer', model: 'tts-1-hd' }, ctx);
|
|
108
|
+
const [url, opts] = mockFetch.mock.calls[0];
|
|
109
|
+
expect(url).toContain('/audio/speech');
|
|
110
|
+
const body = JSON.parse(opts.body);
|
|
111
|
+
expect(body.voice).toBe('shimmer');
|
|
112
|
+
expect(body.model).toBe('tts-1-hd');
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
it('handles API errors', async () => {
|
|
116
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
117
|
+
mockFetch.mockResolvedValueOnce({ ok: false, status: 401, text: async () => 'Unauthorized' });
|
|
118
|
+
const result = await tool.execute({ text: 'Test' }, ctx);
|
|
119
|
+
expect(result.success).toBe(false);
|
|
120
|
+
expect(result.error).toContain('401');
|
|
121
|
+
});
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
describe('ElevenLabs TTS', () => {
|
|
125
|
+
it('synthesizes with default voice (rachel)', async () => {
|
|
126
|
+
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
127
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(100) });
|
|
128
|
+
const result = await tool.execute({ text: 'Hello world' }, ctx);
|
|
129
|
+
expect(result.success).toBe(true);
|
|
130
|
+
expect(result.output!.voice).toBe('rachel');
|
|
131
|
+
expect(result.output!.provider).toBe('elevenlabs');
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
it('uses correct voice ID for named voice (josh)', async () => {
|
|
135
|
+
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
59
136
|
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
60
137
|
await tool.execute({ text: 'Test', voice: 'josh' }, ctx);
|
|
61
138
|
expect(mockFetch).toHaveBeenCalledWith(
|
|
@@ -64,26 +141,38 @@ describe('TextToSpeechTool', () => {
|
|
|
64
141
|
);
|
|
65
142
|
});
|
|
66
143
|
|
|
144
|
+
it('sends ElevenLabs API key header', async () => {
|
|
145
|
+
const tool = new TextToSpeechTool({ openaiApiKey: '', elevenLabsApiKey: 'el-test' });
|
|
146
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
147
|
+
await tool.execute({ text: 'Test' }, ctx);
|
|
148
|
+
const [, opts] = mockFetch.mock.calls[0];
|
|
149
|
+
expect(opts.headers['xi-api-key']).toBe('el-test');
|
|
150
|
+
});
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
describe('common behavior', () => {
|
|
67
154
|
it('truncates text to 5000 chars', async () => {
|
|
155
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
68
156
|
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
69
|
-
const
|
|
70
|
-
const result = await tool.execute({ text: longText }, ctx);
|
|
157
|
+
const result = await tool.execute({ text: 'a'.repeat(6000) }, ctx);
|
|
71
158
|
expect(result.success).toBe(true);
|
|
72
159
|
expect(result.output!.text.length).toBe(5000);
|
|
73
160
|
});
|
|
74
161
|
|
|
75
|
-
it('
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
expect(result.
|
|
162
|
+
it('estimates duration from word count', async () => {
|
|
163
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
164
|
+
mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
|
|
165
|
+
const result = await tool.execute({ text: 'one two three four five' }, ctx);
|
|
166
|
+
expect(result.success).toBe(true);
|
|
167
|
+
expect(result.output!.durationEstimateMs).toBeGreaterThan(0);
|
|
80
168
|
});
|
|
81
169
|
|
|
82
|
-
it('handles network errors', async () => {
|
|
83
|
-
|
|
170
|
+
it('handles network errors gracefully', async () => {
|
|
171
|
+
const tool = new TextToSpeechTool({ openaiApiKey: 'sk-test', elevenLabsApiKey: '' });
|
|
172
|
+
mockFetch.mockRejectedValueOnce(new Error('ECONNREFUSED'));
|
|
84
173
|
const result = await tool.execute({ text: 'Test' }, ctx);
|
|
85
174
|
expect(result.success).toBe(false);
|
|
86
|
-
expect(result.error).toContain('
|
|
175
|
+
expect(result.error).toContain('TTS failed');
|
|
87
176
|
});
|
|
88
177
|
});
|
|
89
178
|
});
|
|
@@ -92,6 +181,9 @@ describe('createExtensionPack', () => {
|
|
|
92
181
|
it('creates pack with correct metadata', () => {
|
|
93
182
|
const pack = createExtensionPack({ options: { elevenLabsApiKey: 'test' }, logger: { info: vi.fn() } });
|
|
94
183
|
expect(pack.name).toBe('@framers/agentos-ext-voice-synthesis');
|
|
184
|
+
expect(pack.version).toBe('2.0.0');
|
|
95
185
|
expect(pack.descriptors).toHaveLength(1);
|
|
186
|
+
expect(pack.descriptors[0].kind).toBe('tool');
|
|
187
|
+
expect(pack.descriptors[0].id).toBe('text_to_speech');
|
|
96
188
|
});
|
|
97
189
|
});
|