@framers/agentos-ext-voice-synthesis 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md ADDED
@@ -0,0 +1,28 @@
1
+ # Voice Synthesis Extension for AgentOS
2
+
3
+ Text-to-speech synthesis using ElevenLabs with 8 predefined voices.
4
+
5
+ ## Installation
6
+
7
+ ```bash
8
+ npm install @framers/agentos-ext-voice-synthesis
9
+ ```
10
+
11
+ ## Configuration
12
+
13
+ Set `ELEVENLABS_API_KEY` environment variable or pass via options.
14
+
15
+ ## Tool: text_to_speech
16
+
17
+ **Input:**
18
+ - `text` (string, required) — Text to convert (max 5000 chars)
19
+ - `voice` (string, default: rachel) — Voice: rachel, domi, bella, antoni, josh, arnold, adam, sam
20
+ - `model` (string, default: eleven_monolingual_v1) — ElevenLabs model
21
+ - `stability` (number, 0-1, default: 0.5) — Voice stability
22
+ - `similarity_boost` (number, 0-1, default: 0.75) — Voice similarity
23
+
24
+ **Output:** Base64-encoded MP3 audio with duration estimate.
25
+
26
+ ## License
27
+
28
+ MIT - Frame.dev
@@ -0,0 +1,27 @@
1
+ /**
2
+ * Voice Synthesis Extension Pack — ElevenLabs TTS for agents.
3
+ */
4
+ import { TextToSpeechTool } from './tools/textToSpeech.js';
5
+ export interface VoiceSynthesisExtensionOptions {
6
+ elevenLabsApiKey?: string;
7
+ priority?: number;
8
+ }
9
+ export declare function createExtensionPack(context: any): {
10
+ name: string;
11
+ version: string;
12
+ descriptors: {
13
+ id: string;
14
+ kind: "tool";
15
+ priority: number;
16
+ payload: TextToSpeechTool;
17
+ requiredSecrets: {
18
+ id: string;
19
+ }[];
20
+ }[];
21
+ onActivate: () => Promise<any>;
22
+ onDeactivate: () => Promise<any>;
23
+ };
24
+ export { TextToSpeechTool };
25
+ export type { TTSInput, TTSOutput } from './tools/textToSpeech.js';
26
+ export default createExtensionPack;
27
+ //# sourceMappingURL=index.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;GAEG;AAEH,OAAO,EAAE,gBAAgB,EAAE,MAAM,yBAAyB,CAAC;AAE3D,MAAM,WAAW,8BAA8B;IAC7C,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,QAAQ,CAAC,EAAE,MAAM,CAAC;CACnB;AAED,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,GAAG;;;;;;;;;;;;;;EAc/C;AAED,OAAO,EAAE,gBAAgB,EAAE,CAAC;AAC5B,YAAY,EAAE,QAAQ,EAAE,SAAS,EAAE,MAAM,yBAAyB,CAAC;AACnE,eAAe,mBAAmB,CAAC"}
package/dist/index.js ADDED
@@ -0,0 +1,21 @@
1
+ /**
2
+ * Voice Synthesis Extension Pack — ElevenLabs TTS for agents.
3
+ */
4
+ import { TextToSpeechTool } from './tools/textToSpeech.js';
5
+ export function createExtensionPack(context) {
6
+ const options = (context.options || {});
7
+ const apiKey = options.elevenLabsApiKey || context.getSecret?.('elevenlabs.apiKey') || process.env.ELEVENLABS_API_KEY;
8
+ const tool = new TextToSpeechTool(apiKey);
9
+ return {
10
+ name: '@framers/agentos-ext-voice-synthesis',
11
+ version: '1.0.0',
12
+ descriptors: [
13
+ { id: 'textToSpeech', kind: 'tool', priority: options.priority || 50, payload: tool, requiredSecrets: [{ id: 'elevenlabs.apiKey' }] },
14
+ ],
15
+ onActivate: async () => context.logger?.info('Voice Synthesis Extension activated'),
16
+ onDeactivate: async () => context.logger?.info('Voice Synthesis Extension deactivated'),
17
+ };
18
+ }
19
+ export { TextToSpeechTool };
20
+ export default createExtensionPack;
21
+ //# sourceMappingURL=index.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;GAEG;AAEH,OAAO,EAAE,gBAAgB,EAAE,MAAM,yBAAyB,CAAC;AAO3D,MAAM,UAAU,mBAAmB,CAAC,OAAY;IAC9C,MAAM,OAAO,GAAG,CAAC,OAAO,CAAC,OAAO,IAAI,EAAE,CAAmC,CAAC;IAC1E,MAAM,MAAM,GAAG,OAAO,CAAC,gBAAgB,IAAI,OAAO,CAAC,SAAS,EAAE,CAAC,mBAAmB,CAAC,IAAI,OAAO,CAAC,GAAG,CAAC,kBAAkB,CAAC;IACtH,MAAM,IAAI,GAAG,IAAI,gBAAgB,CAAC,MAAM,CAAC,CAAC;IAE1C,OAAO;QACL,IAAI,EAAE,sCAAsC;QAC5C,OAAO,EAAE,OAAO;QAChB,WAAW,EAAE;YACX,EAAE,EAAE,EAAE,cAAc,EAAE,IAAI,EAAE,MAAe,EAAE,QAAQ,EAAE,OAAO,CAAC,QAAQ,IAAI,EAAE,EAAE,OAAO,EAAE,IAAI,EAAE,eAAe,EAAE,CAAC,EAAE,EAAE,EAAE,mBAAmB,EAAE,CAAC,EAAE;SAC/I;QACD,UAAU,EAAE,KAAK,IAAI,EAAE,CAAC,OAAO,CAAC,MAAM,EAAE,IAAI,CAAC,qCAAqC,CAAC;QACnF,YAAY,EAAE,KAAK,IAAI,EAAE,CAAC,OAAO,CAAC,MAAM,EAAE,IAAI,CAAC,uCAAuC,CAAC;KACxF,CAAC;AACJ,CAAC;AAED,OAAO,EAAE,gBAAgB,EAAE,CAAC;AAE5B,eAAe,mBAAmB,CAAC"}
@@ -0,0 +1,34 @@
1
+ /**
2
+ * ElevenLabs TTS Tool — text-to-speech synthesis.
3
+ */
4
+ import type { ITool, ToolExecutionContext, ToolExecutionResult, JSONSchemaObject } from '../../../../../../agentos/src/core/tools/ITool.js';
5
+ export interface TTSInput {
6
+ text: string;
7
+ voice?: string;
8
+ model?: string;
9
+ stability?: number;
10
+ similarity_boost?: number;
11
+ }
12
+ export interface TTSOutput {
13
+ text: string;
14
+ voice: string;
15
+ model: string;
16
+ audioBase64: string;
17
+ contentType: string;
18
+ durationEstimateMs: number;
19
+ }
20
+ export declare class TextToSpeechTool implements ITool<TTSInput, TTSOutput> {
21
+ readonly id = "elevenlabs-tts-v1";
22
+ readonly name = "text_to_speech";
23
+ readonly displayName = "Text to Speech";
24
+ readonly description: string;
25
+ readonly category = "media";
26
+ readonly version = "1.0.0";
27
+ readonly hasSideEffects = false;
28
+ readonly inputSchema: JSONSchemaObject;
29
+ readonly requiredCapabilities: string[];
30
+ private apiKey;
31
+ constructor(apiKey?: string);
32
+ execute(args: TTSInput, _context: ToolExecutionContext): Promise<ToolExecutionResult<TTSOutput>>;
33
+ }
34
+ //# sourceMappingURL=textToSpeech.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"textToSpeech.d.ts","sourceRoot":"","sources":["../../src/tools/textToSpeech.ts"],"names":[],"mappings":"AAAA;;GAEG;AAEH,OAAO,KAAK,EAAE,KAAK,EAAE,oBAAoB,EAAE,mBAAmB,EAAE,gBAAgB,EAAE,MAAM,mDAAmD,CAAC;AAE5I,MAAM,WAAW,QAAQ;IACvB,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gBAAgB,CAAC,EAAE,MAAM,CAAC;CAC3B;AAED,MAAM,WAAW,SAAS;IACxB,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;IACd,KAAK,EAAE,MAAM,CAAC;IACd,WAAW,EAAE,MAAM,CAAC;IACpB,WAAW,EAAE,MAAM,CAAC;IACpB,kBAAkB,EAAE,MAAM,CAAC;CAC5B;AAaD,qBAAa,gBAAiB,YAAW,KAAK,CAAC,QAAQ,EAAE,SAAS,CAAC;IACjE,QAAQ,CAAC,EAAE,uBAAuB;IAClC,QAAQ,CAAC,IAAI,oBAAoB;IACjC,QAAQ,CAAC,WAAW,oBAAoB;IACxC,QAAQ,CAAC,WAAW,SAE8C;IAClE,QAAQ,CAAC,QAAQ,WAAW;IAC5B,QAAQ,CAAC,OAAO,WAAW;IAC3B,QAAQ,CAAC,cAAc,SAAS;IAEhC,QAAQ,CAAC,WAAW,EAAE,gBAAgB,CAUpC;IAEF,QAAQ,CAAC,oBAAoB,WAAsB;IAEnD,OAAO,CAAC,MAAM,CAAS;gBAEX,MAAM,CAAC,EAAE,MAAM;IAIrB,OAAO,CAAC,IAAI,EAAE,QAAQ,EAAE,QAAQ,EAAE,oBAAoB,GAAG,OAAO,CAAC,mBAAmB,CAAC,SAAS,CAAC,CAAC;CAoCvG"}
@@ -0,0 +1,73 @@
1
+ /**
2
+ * ElevenLabs TTS Tool — text-to-speech synthesis.
3
+ */
4
+ const VOICES = {
5
+ rachel: '21m00Tcm4TlvDq8ikWAM',
6
+ domi: 'AZnzlk1XvdvUeBnXmlld',
7
+ bella: 'EXAVITQu4vr4xnSDxMaL',
8
+ antoni: 'ErXwobaYiN019PkySvjV',
9
+ josh: 'TxGEqnHWrfWFTfGW9XjX',
10
+ arnold: 'VR6AewLTigWG4xSOukaG',
11
+ adam: 'pNInz6obpgDQGcFmaJgB',
12
+ sam: 'yoZ06aMxZJJ28mfd3POQ',
13
+ };
14
+ export class TextToSpeechTool {
15
+ id = 'elevenlabs-tts-v1';
16
+ name = 'text_to_speech';
17
+ displayName = 'Text to Speech';
18
+ description = 'Convert text to speech using ElevenLabs. Returns base64-encoded MP3 audio. ' +
19
+ 'Voices: rachel, domi, bella, antoni, josh, arnold, adam, sam.';
20
+ category = 'media';
21
+ version = '1.0.0';
22
+ hasSideEffects = false;
23
+ inputSchema = {
24
+ type: 'object',
25
+ properties: {
26
+ text: { type: 'string', description: 'Text to convert. Max 5000 chars.' },
27
+ voice: { type: 'string', default: 'rachel', description: 'Voice name or ID.' },
28
+ model: { type: 'string', default: 'eleven_monolingual_v1' },
29
+ stability: { type: 'number', minimum: 0, maximum: 1, default: 0.5 },
30
+ similarity_boost: { type: 'number', minimum: 0, maximum: 1, default: 0.75 },
31
+ },
32
+ required: ['text'],
33
+ };
34
+ requiredCapabilities = ['capability:tts'];
35
+ apiKey;
36
+ constructor(apiKey) {
37
+ this.apiKey = apiKey || process.env.ELEVENLABS_API_KEY || '';
38
+ }
39
+ async execute(args, _context) {
40
+ if (!this.apiKey)
41
+ return { success: false, error: 'ELEVENLABS_API_KEY not configured.' };
42
+ const text = args.text.slice(0, 5000);
43
+ const voiceId = VOICES[(args.voice || 'rachel').toLowerCase()] || args.voice || VOICES.rachel;
44
+ const model = args.model || 'eleven_monolingual_v1';
45
+ try {
46
+ const response = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`, {
47
+ method: 'POST',
48
+ headers: { 'xi-api-key': this.apiKey, 'Content-Type': 'application/json', Accept: 'audio/mpeg' },
49
+ body: JSON.stringify({
50
+ text,
51
+ model_id: model,
52
+ voice_settings: { stability: args.stability ?? 0.5, similarity_boost: args.similarity_boost ?? 0.75 },
53
+ }),
54
+ });
55
+ if (!response.ok) {
56
+ const err = await response.text();
57
+ return { success: false, error: `ElevenLabs error (${response.status}): ${err}` };
58
+ }
59
+ const buf = await response.arrayBuffer();
60
+ const audioBase64 = Buffer.from(buf).toString('base64');
61
+ const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
62
+ return {
63
+ success: true,
64
+ output: { text, voice: args.voice || 'rachel', model, audioBase64, contentType: 'audio/mpeg', durationEstimateMs },
65
+ contentType: 'audio/mpeg',
66
+ };
67
+ }
68
+ catch (err) {
69
+ return { success: false, error: `TTS failed: ${err.message}` };
70
+ }
71
+ }
72
+ }
73
+ //# sourceMappingURL=textToSpeech.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"textToSpeech.js","sourceRoot":"","sources":["../../src/tools/textToSpeech.ts"],"names":[],"mappings":"AAAA;;GAEG;AAqBH,MAAM,MAAM,GAA2B;IACrC,MAAM,EAAE,sBAAsB;IAC9B,IAAI,EAAE,sBAAsB;IAC5B,KAAK,EAAE,sBAAsB;IAC7B,MAAM,EAAE,sBAAsB;IAC9B,IAAI,EAAE,sBAAsB;IAC5B,MAAM,EAAE,sBAAsB;IAC9B,IAAI,EAAE,sBAAsB;IAC5B,GAAG,EAAE,sBAAsB;CAC5B,CAAC;AAEF,MAAM,OAAO,gBAAgB;IAClB,EAAE,GAAG,mBAAmB,CAAC;IACzB,IAAI,GAAG,gBAAgB,CAAC;IACxB,WAAW,GAAG,gBAAgB,CAAC;IAC/B,WAAW,GAClB,6EAA6E;QAC7E,+DAA+D,CAAC;IACzD,QAAQ,GAAG,OAAO,CAAC;IACnB,OAAO,GAAG,OAAO,CAAC;IAClB,cAAc,GAAG,KAAK,CAAC;IAEvB,WAAW,GAAqB;QACvC,IAAI,EAAE,QAAQ;QACd,UAAU,EAAE;YACV,IAAI,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,WAAW,EAAE,kCAAkC,EAAE;YACzE,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,QAAQ,EAAE,WAAW,EAAE,mBAAmB,EAAE;YAC9E,KAAK,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,uBAAuB,EAAE;YAC3D,SAAS,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,OAAO,EAAE,GAAG,EAAE;YACnE,gBAAgB,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,OAAO,EAAE,IAAI,EAAE;SAC5E;QACD,QAAQ,EAAE,CAAC,MAAM,CAAC;KACnB,CAAC;IAEO,oBAAoB,GAAG,CAAC,gBAAgB,CAAC,CAAC;IAE3C,MAAM,CAAS;IAEvB,YAAY,MAAe;QACzB,IAAI,CAAC,MAAM,GAAG,MAAM,IAAI,OAAO,CAAC,GAAG,CAAC,kBAAkB,IAAI,EAAE,CAAC;IAC/D,CAAC;IAED,KAAK,CAAC,OAAO,CAAC,IAAc,EAAE,QAA8B;QAC1D,IAAI,CAAC,IAAI,CAAC,MAAM;YAAE,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,oCAAoC,EAAE,CAAC;QAEzF,MAAM,IAAI,GAAG,IAAI,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,CAAC;QACtC,MAAM,OAAO,GAAG,MAAM,CAAC,CAAC,IAAI,CAAC,KAAK,IAAI,QAAQ,CAAC,CAAC,WAAW,EAAE,CAAC,IAAI,IAAI,CAAC,KAAK,IAAI,MAAM,CAAC,MAAM,CAAC;QAC9F,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,uBAAuB,CAAC;QAEpD,IAAI,CAAC;YACH,MAAM,QAAQ,GAAG,MAAM,KAAK,CAAC,+CAA+C,OAAO,EAAE,EAAE;gBACrF,MAAM,EAAE,MAAM;gBACd,OAAO,EAAE,EAAE,YAAY,EAAE,IAAI,CAAC,MAAM,EAAE,cAAc,EAAE,kBAAkB,EAAE,MAAM,EAAE,YAAY,EAAE;gBAChG,IAAI,EAAE,IAAI,CAAC,SAAS,CAAC;oBACnB,IAAI;oBACJ,QAAQ,EAAE,KAAK;oBACf,cAAc,EAAE,EAAE,SAAS,EAAE,IAAI,CAAC,SAAS,IAAI,GAAG,EAAE,gBAAgB,EAAE,IAAI,CAAC,gBAAgB,IAAI,IAAI,EAAE;iBACtG,CAAC;aACH,CAAC,CAAC;YAEH,IAAI,CAAC,QAAQ,CAAC,EAAE,EAAE,CAAC;gBACjB,MAAM,GAAG,GAAG,MAAM,QAAQ,CAAC,IAAI,EAAE,CAAC;gBAClC,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,qBAAqB,QAAQ,CAAC,MAAM,MAAM,GAAG,EAAE,EAAE,CAAC;YACpF,CAAC;YAED,MAAM,GAAG,GAAG,MAAM,QAAQ,CAAC,WAAW,EAAE,CAAC;YACzC,MAAM,WAAW,GAAG,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,QAAQ,CAAC,CAAC;YACxD,MAAM,kBAAkB,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,MAAM,GAAG,GAAG,CAAC,GAAG,EAAE,GAAG,IAAI,CAAC,CAAC;YAEpF,OAAO;gBACL,OAAO,EAAE,IAAI;gBACb,MAAM,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,IAAI,QAAQ,EAAE,KAAK,EAAE,WAAW,EAAE,WAAW,EAAE,YAAY,EAAE,kBAAkB,EAAE;gBAClH,WAAW,EAAE,YAAY;aAC1B,CAAC;QACJ,CAAC;QAAC,OAAO,GAAQ,EAAE,CAAC;YAClB,OAAO,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,eAAe,GAAG,CAAC,OAAO,EAAE,EAAE,CAAC;QACjE,CAAC;IACH,CAAC;CACF"}
package/manifest.json ADDED
@@ -0,0 +1,20 @@
1
+ {
2
+ "$schema": "https://agentos.sh/schemas/extension-manifest-v1.json",
3
+ "id": "com.framers.media.voice-synthesis",
4
+ "name": "Voice Synthesis Extension",
5
+ "version": "1.0.0",
6
+ "description": "Text-to-speech synthesis via ElevenLabs for generating agent voice audio",
7
+ "author": { "name": "Frame.dev" },
8
+ "license": "MIT",
9
+ "keywords": ["tts", "voice", "speech", "elevenlabs", "audio"],
10
+ "agentosVersion": "^2.0.0",
11
+ "categories": ["media", "communication"],
12
+ "extensions": [
13
+ { "kind": "tool", "id": "textToSpeech", "displayName": "Text to Speech", "entry": "./src/tools/textToSpeech.ts" }
14
+ ],
15
+ "configuration": {
16
+ "properties": {
17
+ "elevenlabs.apiKey": { "type": "string", "secret": true }
18
+ }
19
+ }
20
+ }
package/package.json ADDED
@@ -0,0 +1,38 @@
1
+ {
2
+ "name": "@framers/agentos-ext-voice-synthesis",
3
+ "version": "1.0.0",
4
+ "description": "ElevenLabs text-to-speech synthesis tool for AgentOS",
5
+ "main": "dist/index.js",
6
+ "types": "dist/index.d.ts",
7
+ "type": "module",
8
+ "sideEffects": false,
9
+ "exports": {
10
+ ".": { "import": "./dist/index.js", "types": "./dist/index.d.ts" },
11
+ "./manifest.json": "./manifest.json",
12
+ "./package.json": "./package.json"
13
+ },
14
+ "scripts": {
15
+ "build": "tsc",
16
+ "test": "vitest run",
17
+ "test:watch": "vitest",
18
+ "test:coverage": "vitest run --coverage",
19
+ "lint": "eslint src --ext .ts",
20
+ "typecheck": "tsc --noEmit",
21
+ "clean": "rimraf dist"
22
+ },
23
+ "keywords": ["agentos", "extension", "tts", "voice", "elevenlabs", "speech", "media"],
24
+ "author": { "name": "Framers AI", "email": "team@frame.dev", "url": "https://frame.dev" },
25
+ "contributors": [{ "name": "Johnny Dunn", "email": "johnnyfived@protonmail.com", "url": "https://github.com/jddunn" }],
26
+ "license": "MIT",
27
+ "repository": { "type": "git", "url": "https://github.com/framersai/agentos-extensions.git", "directory": "registry/curated/media/voice-synthesis" },
28
+ "publishConfig": { "access": "public" },
29
+ "peerDependencies": { "@framers/agentos": "^0.1.0" },
30
+ "devDependencies": {
31
+ "@framers/agentos": "^0.1.0",
32
+ "@types/node": "^20.12.12",
33
+ "@vitest/coverage-v8": "^1.6.0",
34
+ "rimraf": "^5.0.7",
35
+ "typescript": "^5.4.5",
36
+ "vitest": "^1.6.0"
37
+ }
38
+ }
package/src/index.ts ADDED
@@ -0,0 +1,30 @@
1
+ /**
2
+ * Voice Synthesis Extension Pack — ElevenLabs TTS for agents.
3
+ */
4
+
5
+ import { TextToSpeechTool } from './tools/textToSpeech.js';
6
+
7
+ export interface VoiceSynthesisExtensionOptions {
8
+ elevenLabsApiKey?: string;
9
+ priority?: number;
10
+ }
11
+
12
+ export function createExtensionPack(context: any) {
13
+ const options = (context.options || {}) as VoiceSynthesisExtensionOptions;
14
+ const apiKey = options.elevenLabsApiKey || context.getSecret?.('elevenlabs.apiKey') || process.env.ELEVENLABS_API_KEY;
15
+ const tool = new TextToSpeechTool(apiKey);
16
+
17
+ return {
18
+ name: '@framers/agentos-ext-voice-synthesis',
19
+ version: '1.0.0',
20
+ descriptors: [
21
+ { id: 'textToSpeech', kind: 'tool' as const, priority: options.priority || 50, payload: tool, requiredSecrets: [{ id: 'elevenlabs.apiKey' }] },
22
+ ],
23
+ onActivate: async () => context.logger?.info('Voice Synthesis Extension activated'),
24
+ onDeactivate: async () => context.logger?.info('Voice Synthesis Extension deactivated'),
25
+ };
26
+ }
27
+
28
+ export { TextToSpeechTool };
29
+ export type { TTSInput, TTSOutput } from './tools/textToSpeech.js';
30
+ export default createExtensionPack;
@@ -0,0 +1,102 @@
1
+ /**
2
+ * ElevenLabs TTS Tool — text-to-speech synthesis.
3
+ */
4
+
5
+ import type { ITool, ToolExecutionContext, ToolExecutionResult, JSONSchemaObject } from '../../../../../../agentos/src/core/tools/ITool.js';
6
+
7
+ export interface TTSInput {
8
+ text: string;
9
+ voice?: string;
10
+ model?: string;
11
+ stability?: number;
12
+ similarity_boost?: number;
13
+ }
14
+
15
+ export interface TTSOutput {
16
+ text: string;
17
+ voice: string;
18
+ model: string;
19
+ audioBase64: string;
20
+ contentType: string;
21
+ durationEstimateMs: number;
22
+ }
23
+
24
+ const VOICES: Record<string, string> = {
25
+ rachel: '21m00Tcm4TlvDq8ikWAM',
26
+ domi: 'AZnzlk1XvdvUeBnXmlld',
27
+ bella: 'EXAVITQu4vr4xnSDxMaL',
28
+ antoni: 'ErXwobaYiN019PkySvjV',
29
+ josh: 'TxGEqnHWrfWFTfGW9XjX',
30
+ arnold: 'VR6AewLTigWG4xSOukaG',
31
+ adam: 'pNInz6obpgDQGcFmaJgB',
32
+ sam: 'yoZ06aMxZJJ28mfd3POQ',
33
+ };
34
+
35
+ export class TextToSpeechTool implements ITool<TTSInput, TTSOutput> {
36
+ readonly id = 'elevenlabs-tts-v1';
37
+ readonly name = 'text_to_speech';
38
+ readonly displayName = 'Text to Speech';
39
+ readonly description =
40
+ 'Convert text to speech using ElevenLabs. Returns base64-encoded MP3 audio. ' +
41
+ 'Voices: rachel, domi, bella, antoni, josh, arnold, adam, sam.';
42
+ readonly category = 'media';
43
+ readonly version = '1.0.0';
44
+ readonly hasSideEffects = false;
45
+
46
+ readonly inputSchema: JSONSchemaObject = {
47
+ type: 'object',
48
+ properties: {
49
+ text: { type: 'string', description: 'Text to convert. Max 5000 chars.' },
50
+ voice: { type: 'string', default: 'rachel', description: 'Voice name or ID.' },
51
+ model: { type: 'string', default: 'eleven_monolingual_v1' },
52
+ stability: { type: 'number', minimum: 0, maximum: 1, default: 0.5 },
53
+ similarity_boost: { type: 'number', minimum: 0, maximum: 1, default: 0.75 },
54
+ },
55
+ required: ['text'],
56
+ };
57
+
58
+ readonly requiredCapabilities = ['capability:tts'];
59
+
60
+ private apiKey: string;
61
+
62
+ constructor(apiKey?: string) {
63
+ this.apiKey = apiKey || process.env.ELEVENLABS_API_KEY || '';
64
+ }
65
+
66
+ async execute(args: TTSInput, _context: ToolExecutionContext): Promise<ToolExecutionResult<TTSOutput>> {
67
+ if (!this.apiKey) return { success: false, error: 'ELEVENLABS_API_KEY not configured.' };
68
+
69
+ const text = args.text.slice(0, 5000);
70
+ const voiceId = VOICES[(args.voice || 'rachel').toLowerCase()] || args.voice || VOICES.rachel;
71
+ const model = args.model || 'eleven_monolingual_v1';
72
+
73
+ try {
74
+ const response = await fetch(`https://api.elevenlabs.io/v1/text-to-speech/${voiceId}`, {
75
+ method: 'POST',
76
+ headers: { 'xi-api-key': this.apiKey, 'Content-Type': 'application/json', Accept: 'audio/mpeg' },
77
+ body: JSON.stringify({
78
+ text,
79
+ model_id: model,
80
+ voice_settings: { stability: args.stability ?? 0.5, similarity_boost: args.similarity_boost ?? 0.75 },
81
+ }),
82
+ });
83
+
84
+ if (!response.ok) {
85
+ const err = await response.text();
86
+ return { success: false, error: `ElevenLabs error (${response.status}): ${err}` };
87
+ }
88
+
89
+ const buf = await response.arrayBuffer();
90
+ const audioBase64 = Buffer.from(buf).toString('base64');
91
+ const durationEstimateMs = Math.round((text.split(/\s+/).length / 150) * 60 * 1000);
92
+
93
+ return {
94
+ success: true,
95
+ output: { text, voice: args.voice || 'rachel', model, audioBase64, contentType: 'audio/mpeg', durationEstimateMs },
96
+ contentType: 'audio/mpeg',
97
+ };
98
+ } catch (err: any) {
99
+ return { success: false, error: `TTS failed: ${err.message}` };
100
+ }
101
+ }
102
+ }
@@ -0,0 +1,97 @@
1
+ import { describe, it, expect, vi, beforeEach } from 'vitest';
2
+
3
+ const mockFetch = vi.fn();
4
+ vi.stubGlobal('fetch', mockFetch);
5
+
6
+ const { TextToSpeechTool } = await import('../src/tools/textToSpeech.js');
7
+ const { createExtensionPack } = await import('../src/index.js');
8
+
9
+ describe('TextToSpeechTool', () => {
10
+ let tool: InstanceType<typeof TextToSpeechTool>;
11
+
12
+ beforeEach(() => {
13
+ vi.clearAllMocks();
14
+ tool = new TextToSpeechTool('test-api-key');
15
+ });
16
+
17
+ describe('metadata', () => {
18
+ it('has correct id and name', () => {
19
+ expect(tool.id).toBe('elevenlabs-tts-v1');
20
+ expect(tool.name).toBe('text_to_speech');
21
+ });
22
+
23
+ it('has valid input schema', () => {
24
+ expect(tool.inputSchema.type).toBe('object');
25
+ expect(tool.inputSchema.required).toContain('text');
26
+ });
27
+
28
+ it('has no side effects', () => {
29
+ expect(tool.hasSideEffects).toBe(false);
30
+ });
31
+ });
32
+
33
+ describe('execute', () => {
34
+ const ctx = {} as any;
35
+
36
+ it('returns error when API key is missing', async () => {
37
+ const noKeyTool = new TextToSpeechTool('');
38
+ const result = await noKeyTool.execute({ text: 'Hello' }, ctx);
39
+ expect(result.success).toBe(false);
40
+ expect(result.error).toContain('ELEVENLABS_API_KEY');
41
+ });
42
+
43
+ it('synthesizes speech successfully', async () => {
44
+ const mockAudio = new ArrayBuffer(100);
45
+ mockFetch.mockResolvedValueOnce({
46
+ ok: true,
47
+ arrayBuffer: async () => mockAudio,
48
+ });
49
+
50
+ const result = await tool.execute({ text: 'Hello world' }, ctx);
51
+ expect(result.success).toBe(true);
52
+ expect(result.output!.voice).toBe('rachel');
53
+ expect(result.output!.contentType).toBe('audio/mpeg');
54
+ expect(result.output!.audioBase64).toBeTruthy();
55
+ expect(result.output!.durationEstimateMs).toBeGreaterThan(0);
56
+ });
57
+
58
+ it('uses correct voice ID for named voice', async () => {
59
+ mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
60
+ await tool.execute({ text: 'Test', voice: 'josh' }, ctx);
61
+ expect(mockFetch).toHaveBeenCalledWith(
62
+ expect.stringContaining('TxGEqnHWrfWFTfGW9XjX'),
63
+ expect.any(Object)
64
+ );
65
+ });
66
+
67
+ it('truncates text to 5000 chars', async () => {
68
+ mockFetch.mockResolvedValueOnce({ ok: true, arrayBuffer: async () => new ArrayBuffer(10) });
69
+ const longText = 'a'.repeat(6000);
70
+ const result = await tool.execute({ text: longText }, ctx);
71
+ expect(result.success).toBe(true);
72
+ expect(result.output!.text.length).toBe(5000);
73
+ });
74
+
75
+ it('handles API errors', async () => {
76
+ mockFetch.mockResolvedValueOnce({ ok: false, status: 401, text: async () => 'Unauthorized' });
77
+ const result = await tool.execute({ text: 'Test' }, ctx);
78
+ expect(result.success).toBe(false);
79
+ expect(result.error).toContain('401');
80
+ });
81
+
82
+ it('handles network errors', async () => {
83
+ mockFetch.mockRejectedValueOnce(new Error('Timeout'));
84
+ const result = await tool.execute({ text: 'Test' }, ctx);
85
+ expect(result.success).toBe(false);
86
+ expect(result.error).toContain('Timeout');
87
+ });
88
+ });
89
+ });
90
+
91
+ describe('createExtensionPack', () => {
92
+ it('creates pack with correct metadata', () => {
93
+ const pack = createExtensionPack({ options: { elevenLabsApiKey: 'test' }, logger: { info: vi.fn() } });
94
+ expect(pack.name).toBe('@framers/agentos-ext-voice-synthesis');
95
+ expect(pack.descriptors).toHaveLength(1);
96
+ });
97
+ });
package/tsconfig.json ADDED
@@ -0,0 +1,22 @@
1
+ {
2
+ "compilerOptions": {
3
+ "target": "ES2022",
4
+ "module": "ESNext",
5
+ "lib": ["ES2022"],
6
+ "outDir": "./dist",
7
+ "rootDir": "./src",
8
+ "strict": true,
9
+ "esModuleInterop": true,
10
+ "skipLibCheck": true,
11
+ "noEmitOnError": false,
12
+ "forceConsistentCasingInFileNames": true,
13
+ "declaration": true,
14
+ "declarationMap": true,
15
+ "sourceMap": true,
16
+ "resolveJsonModule": true,
17
+ "moduleResolution": "Bundler",
18
+ "types": ["node"]
19
+ },
20
+ "include": ["src/**/*"],
21
+ "exclude": ["node_modules", "dist", "test"]
22
+ }
@@ -0,0 +1,10 @@
1
+ import { defineConfig } from 'vitest/config';
2
+
3
+ export default defineConfig({
4
+ test: {
5
+ globals: true,
6
+ environment: 'node',
7
+ include: ['test/**/*.spec.ts'],
8
+ testTimeout: 10000,
9
+ },
10
+ });