@tanstack/ai-elevenlabs 0.1.8 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +93 -0
- package/dist/esm/adapters/audio.js +106 -0
- package/dist/esm/adapters/audio.js.map +1 -0
- package/dist/esm/adapters/speech.d.ts +83 -0
- package/dist/esm/adapters/speech.js +109 -0
- package/dist/esm/adapters/speech.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +79 -0
- package/dist/esm/adapters/transcription.js +142 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/index.d.ts +5 -0
- package/dist/esm/index.js +21 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +42 -0
- package/dist/esm/model-meta.js +32 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/realtime/adapter.d.ts +1 -1
- package/dist/esm/realtime/adapter.js +1 -1
- package/dist/esm/realtime/adapter.js.map +1 -1
- package/dist/esm/realtime/token.d.ts +11 -7
- package/dist/esm/realtime/token.js +8 -32
- package/dist/esm/realtime/token.js.map +1 -1
- package/dist/esm/realtime/types.d.ts +5 -2
- package/dist/esm/utils/client.d.ts +62 -0
- package/dist/esm/utils/client.js +124 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/utils/index.d.ts +1 -0
- package/package.json +15 -7
- package/src/adapters/audio.ts +257 -0
- package/src/adapters/speech.ts +240 -0
- package/src/adapters/transcription.ts +338 -0
- package/src/index.ts +64 -0
- package/src/model-meta.ts +79 -0
- package/src/realtime/adapter.ts +3 -3
- package/src/realtime/token.ts +21 -56
- package/src/realtime/types.ts +5 -2
- package/src/utils/client.ts +196 -0
- package/src/utils/index.ts +10 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai-elevenlabs",
|
|
3
|
-
"version": "0.1
|
|
3
|
+
"version": "0.2.1",
|
|
4
4
|
"description": "ElevenLabs adapter for TanStack AI realtime voice",
|
|
5
5
|
"author": "",
|
|
6
6
|
"license": "MIT",
|
|
@@ -15,7 +15,14 @@
|
|
|
15
15
|
"voice",
|
|
16
16
|
"realtime",
|
|
17
17
|
"tanstack",
|
|
18
|
-
"adapter"
|
|
18
|
+
"adapter",
|
|
19
|
+
"tts",
|
|
20
|
+
"text-to-speech",
|
|
21
|
+
"audio-generation",
|
|
22
|
+
"music",
|
|
23
|
+
"sound-effects",
|
|
24
|
+
"transcription",
|
|
25
|
+
"speech-to-text"
|
|
19
26
|
],
|
|
20
27
|
"type": "module",
|
|
21
28
|
"module": "./dist/esm/index.js",
|
|
@@ -31,16 +38,17 @@
|
|
|
31
38
|
"src"
|
|
32
39
|
],
|
|
33
40
|
"dependencies": {
|
|
34
|
-
"@
|
|
41
|
+
"@elevenlabs/client": "^1.3.1",
|
|
42
|
+
"@elevenlabs/elevenlabs-js": "^2.44.0"
|
|
35
43
|
},
|
|
36
44
|
"peerDependencies": {
|
|
37
|
-
"@tanstack/ai": "^0.
|
|
38
|
-
"@tanstack/ai-client": "^0.
|
|
45
|
+
"@tanstack/ai": "^0.15.0",
|
|
46
|
+
"@tanstack/ai-client": "^0.9.0"
|
|
39
47
|
},
|
|
40
48
|
"devDependencies": {
|
|
41
49
|
"@vitest/coverage-v8": "4.0.14",
|
|
42
|
-
"@tanstack/ai": "0.
|
|
43
|
-
"@tanstack/ai-client": "0.
|
|
50
|
+
"@tanstack/ai": "0.15.0",
|
|
51
|
+
"@tanstack/ai-client": "0.9.0"
|
|
44
52
|
},
|
|
45
53
|
"scripts": {
|
|
46
54
|
"build": "vite build",
|
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
import { BaseAudioAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import {
|
|
3
|
+
arrayBufferToBase64,
|
|
4
|
+
createElevenLabsClient,
|
|
5
|
+
generateId,
|
|
6
|
+
parseOutputFormat,
|
|
7
|
+
readStreamToArrayBuffer,
|
|
8
|
+
} from '../utils/client'
|
|
9
|
+
import {
|
|
10
|
+
isElevenLabsMusicModel,
|
|
11
|
+
isElevenLabsSoundEffectsModel,
|
|
12
|
+
} from '../model-meta'
|
|
13
|
+
import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
|
|
14
|
+
import type {
|
|
15
|
+
AudioGenerationOptions,
|
|
16
|
+
AudioGenerationResult,
|
|
17
|
+
} from '@tanstack/ai'
|
|
18
|
+
import type { ElevenLabsClientConfig } from '../utils/client'
|
|
19
|
+
import type {
|
|
20
|
+
ElevenLabsAudioModel,
|
|
21
|
+
ElevenLabsMusicModel,
|
|
22
|
+
ElevenLabsOutputFormat,
|
|
23
|
+
ElevenLabsSoundEffectsModel,
|
|
24
|
+
} from '../model-meta'
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Structured composition plan for ElevenLabs music generation. Mutually
|
|
28
|
+
* exclusive with a free-form `prompt` on the `generateAudio()` call — when
|
|
29
|
+
* supplied, `prompt` is ignored by ElevenLabs.
|
|
30
|
+
*
|
|
31
|
+
* We mirror the SDK's camelCase naming. Lengths are in milliseconds.
|
|
32
|
+
* @see https://elevenlabs.io/docs/api-reference/music/compose
|
|
33
|
+
*/
|
|
34
|
+
export interface ElevenLabsMusicCompositionPlan {
|
|
35
|
+
/** Positive global style descriptors (mood, instruments, tempo, …). */
|
|
36
|
+
positiveGlobalStyles?: Array<string>
|
|
37
|
+
/** Negative global style descriptors — styles to avoid. */
|
|
38
|
+
negativeGlobalStyles?: Array<string>
|
|
39
|
+
/** Section definitions (verse/chorus/bridge/…) with local style hints. */
|
|
40
|
+
sections?: Array<{
|
|
41
|
+
sectionName: string
|
|
42
|
+
positiveLocalStyles?: Array<string>
|
|
43
|
+
negativeLocalStyles?: Array<string>
|
|
44
|
+
durationMs?: number
|
|
45
|
+
lines?: Array<string>
|
|
46
|
+
}>
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Provider options common to all ElevenLabs audio endpoints.
|
|
51
|
+
*/
|
|
52
|
+
interface CommonAudioOptions {
|
|
53
|
+
/** Output audio format. Defaults to `mp3_44100_128`. */
|
|
54
|
+
outputFormat?: ElevenLabsOutputFormat
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Provider options for music generation (`music_v1`).
|
|
59
|
+
*/
|
|
60
|
+
export interface ElevenLabsMusicProviderOptions extends CommonAudioOptions {
|
|
61
|
+
/** Structured composition plan. Mutually exclusive with `prompt`/`duration`. */
|
|
62
|
+
compositionPlan?: ElevenLabsMusicCompositionPlan
|
|
63
|
+
/** Deterministic sampling seed (incompatible with `prompt`). */
|
|
64
|
+
seed?: number
|
|
65
|
+
/** Force the output to be purely instrumental (prompt-mode only). */
|
|
66
|
+
forceInstrumental?: boolean
|
|
67
|
+
/** Strictly respect section durations in `compositionPlan`. */
|
|
68
|
+
respectSectionsDurations?: boolean
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Provider options for sound-effect generation (`eleven_text_to_sound_v*`).
|
|
73
|
+
*/
|
|
74
|
+
export interface ElevenLabsSoundEffectsProviderOptions extends CommonAudioOptions {
|
|
75
|
+
/** Prompt influence, 0..1. Default 0.3. Higher = more prompt adherence. */
|
|
76
|
+
promptInfluence?: number
|
|
77
|
+
/** Generate a loopable SFX (v2 only). */
|
|
78
|
+
loop?: boolean
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Union of per-model provider options. We keep both branches on one type so
|
|
83
|
+
* the adapter stays tree-shakeable; callers narrow by model at the factory.
|
|
84
|
+
*/
|
|
85
|
+
export type ElevenLabsAudioProviderOptions =
|
|
86
|
+
| (ElevenLabsMusicProviderOptions & ElevenLabsSoundEffectsProviderOptions)
|
|
87
|
+
| ElevenLabsMusicProviderOptions
|
|
88
|
+
| ElevenLabsSoundEffectsProviderOptions
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* ElevenLabs audio generation adapter. Dispatches to music or SFX endpoints
|
|
92
|
+
* based on the model id. Music → `client.music.compose`, SFX →
|
|
93
|
+
* `client.textToSoundEffects.convert`.
|
|
94
|
+
*
|
|
95
|
+
* @example
|
|
96
|
+
* ```ts
|
|
97
|
+
* const music = elevenlabsAudio('music_v1')
|
|
98
|
+
* await generateAudio({ adapter: music, prompt: 'lo-fi beat', duration: 15 })
|
|
99
|
+
*
|
|
100
|
+
* const sfx = elevenlabsAudio('eleven_text_to_sound_v2')
|
|
101
|
+
* await generateAudio({ adapter: sfx, prompt: 'glass shattering', duration: 3 })
|
|
102
|
+
* ```
|
|
103
|
+
*/
|
|
104
|
+
export class ElevenLabsAudioAdapter<
|
|
105
|
+
TModel extends ElevenLabsAudioModel,
|
|
106
|
+
> extends BaseAudioAdapter<TModel, ElevenLabsAudioProviderOptions> {
|
|
107
|
+
readonly name = 'elevenlabs' as const
|
|
108
|
+
|
|
109
|
+
private client: ElevenLabsClient
|
|
110
|
+
|
|
111
|
+
constructor(model: TModel, config?: ElevenLabsClientConfig) {
|
|
112
|
+
super(model, config ?? {})
|
|
113
|
+
this.client = createElevenLabsClient(config)
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
async generateAudio(
|
|
117
|
+
options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,
|
|
118
|
+
): Promise<AudioGenerationResult> {
|
|
119
|
+
const { logger } = options
|
|
120
|
+
logger.request(
|
|
121
|
+
`activity=generateAudio provider=elevenlabs model=${this.model}`,
|
|
122
|
+
{ provider: 'elevenlabs', model: this.model },
|
|
123
|
+
)
|
|
124
|
+
try {
|
|
125
|
+
if (isElevenLabsMusicModel(this.model)) {
|
|
126
|
+
return await this.runMusic(options)
|
|
127
|
+
}
|
|
128
|
+
if (isElevenLabsSoundEffectsModel(this.model)) {
|
|
129
|
+
return await this.runSoundEffects(options)
|
|
130
|
+
}
|
|
131
|
+
throw new Error(
|
|
132
|
+
`Unsupported ElevenLabs audio model "${this.model}". Expected one of: music_v1, eleven_text_to_sound_v2, eleven_text_to_sound_v1.`,
|
|
133
|
+
)
|
|
134
|
+
} catch (error) {
|
|
135
|
+
logger.errors('elevenlabs.generateAudio fatal', {
|
|
136
|
+
error,
|
|
137
|
+
source: 'elevenlabs.generateAudio',
|
|
138
|
+
})
|
|
139
|
+
throw error
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
private async runMusic(
|
|
144
|
+
options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,
|
|
145
|
+
): Promise<AudioGenerationResult> {
|
|
146
|
+
// Gated by isElevenLabsMusicModel() in generateAudio().
|
|
147
|
+
const modelId = this.model as ElevenLabsMusicModel
|
|
148
|
+
const music = (options.modelOptions ?? {}) as ElevenLabsMusicProviderOptions
|
|
149
|
+
const outputFormat = music.outputFormat
|
|
150
|
+
|
|
151
|
+
const stream = await this.client.music.compose({
|
|
152
|
+
modelId,
|
|
153
|
+
...(options.prompt && !music.compositionPlan
|
|
154
|
+
? { prompt: options.prompt }
|
|
155
|
+
: {}),
|
|
156
|
+
...(music.compositionPlan
|
|
157
|
+
? { compositionPlan: toMusicPrompt(music.compositionPlan) }
|
|
158
|
+
: {}),
|
|
159
|
+
...(options.duration != null && !music.compositionPlan
|
|
160
|
+
? { musicLengthMs: Math.round(options.duration * 1000) }
|
|
161
|
+
: {}),
|
|
162
|
+
...(outputFormat ? { outputFormat } : {}),
|
|
163
|
+
...(music.seed != null ? { seed: music.seed } : {}),
|
|
164
|
+
...(music.forceInstrumental != null
|
|
165
|
+
? { forceInstrumental: music.forceInstrumental }
|
|
166
|
+
: {}),
|
|
167
|
+
...(music.respectSectionsDurations != null
|
|
168
|
+
? { respectSectionsDurations: music.respectSectionsDurations }
|
|
169
|
+
: {}),
|
|
170
|
+
})
|
|
171
|
+
|
|
172
|
+
return this.finalize(stream, outputFormat, options.duration)
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
private async runSoundEffects(
|
|
176
|
+
options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,
|
|
177
|
+
): Promise<AudioGenerationResult> {
|
|
178
|
+
// Gated by isElevenLabsSoundEffectsModel() in generateAudio().
|
|
179
|
+
const modelId = this.model as ElevenLabsSoundEffectsModel
|
|
180
|
+
const sfx = (options.modelOptions ??
|
|
181
|
+
{}) as ElevenLabsSoundEffectsProviderOptions
|
|
182
|
+
const outputFormat = sfx.outputFormat
|
|
183
|
+
|
|
184
|
+
const stream = await this.client.textToSoundEffects.convert({
|
|
185
|
+
text: options.prompt,
|
|
186
|
+
modelId,
|
|
187
|
+
...(options.duration != null
|
|
188
|
+
? { durationSeconds: options.duration }
|
|
189
|
+
: {}),
|
|
190
|
+
...(outputFormat ? { outputFormat } : {}),
|
|
191
|
+
...(sfx.promptInfluence != null
|
|
192
|
+
? { promptInfluence: sfx.promptInfluence }
|
|
193
|
+
: {}),
|
|
194
|
+
...(sfx.loop != null ? { loop: sfx.loop } : {}),
|
|
195
|
+
})
|
|
196
|
+
|
|
197
|
+
return this.finalize(stream, outputFormat, options.duration)
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
private async finalize(
|
|
201
|
+
stream: ReadableStream<Uint8Array>,
|
|
202
|
+
outputFormat: ElevenLabsOutputFormat | undefined,
|
|
203
|
+
duration: number | undefined,
|
|
204
|
+
): Promise<AudioGenerationResult> {
|
|
205
|
+
const buffer = await readStreamToArrayBuffer(stream)
|
|
206
|
+
const base64 = arrayBufferToBase64(buffer)
|
|
207
|
+
const { contentType } = parseOutputFormat(outputFormat)
|
|
208
|
+
return {
|
|
209
|
+
id: generateId(this.name),
|
|
210
|
+
model: this.model,
|
|
211
|
+
audio: {
|
|
212
|
+
b64Json: base64,
|
|
213
|
+
contentType,
|
|
214
|
+
...(duration != null ? { duration } : {}),
|
|
215
|
+
},
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
protected override generateId(): string {
|
|
220
|
+
return generateId(this.name)
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
function toMusicPrompt(plan: ElevenLabsMusicCompositionPlan) {
|
|
225
|
+
return {
|
|
226
|
+
positiveGlobalStyles: plan.positiveGlobalStyles ?? [],
|
|
227
|
+
negativeGlobalStyles: plan.negativeGlobalStyles ?? [],
|
|
228
|
+
sections: (plan.sections ?? []).map((section) => ({
|
|
229
|
+
sectionName: section.sectionName,
|
|
230
|
+
positiveLocalStyles: section.positiveLocalStyles ?? [],
|
|
231
|
+
negativeLocalStyles: section.negativeLocalStyles ?? [],
|
|
232
|
+
durationMs: section.durationMs ?? 10000,
|
|
233
|
+
lines: section.lines ?? [],
|
|
234
|
+
})),
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* Create an ElevenLabs audio adapter using `ELEVENLABS_API_KEY` from env.
|
|
240
|
+
*/
|
|
241
|
+
export function elevenlabsAudio<TModel extends ElevenLabsAudioModel>(
|
|
242
|
+
model: TModel,
|
|
243
|
+
config?: ElevenLabsClientConfig,
|
|
244
|
+
): ElevenLabsAudioAdapter<TModel> {
|
|
245
|
+
return new ElevenLabsAudioAdapter(model, config)
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Create an ElevenLabs audio adapter with an explicit API key.
|
|
250
|
+
*/
|
|
251
|
+
export function createElevenLabsAudio<TModel extends ElevenLabsAudioModel>(
|
|
252
|
+
model: TModel,
|
|
253
|
+
apiKey: string,
|
|
254
|
+
config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
|
|
255
|
+
): ElevenLabsAudioAdapter<TModel> {
|
|
256
|
+
return new ElevenLabsAudioAdapter(model, { apiKey, ...config })
|
|
257
|
+
}
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
import { BaseTTSAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import {
|
|
3
|
+
arrayBufferToBase64,
|
|
4
|
+
createElevenLabsClient,
|
|
5
|
+
generateId,
|
|
6
|
+
parseOutputFormat,
|
|
7
|
+
readStreamToArrayBuffer,
|
|
8
|
+
} from '../utils/client'
|
|
9
|
+
import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
|
|
10
|
+
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
11
|
+
import type { ElevenLabsClientConfig } from '../utils/client'
|
|
12
|
+
import type { ElevenLabsOutputFormat, ElevenLabsTTSModel } from '../model-meta'
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* ElevenLabs voice settings overrides. All fields are optional — omitted
|
|
16
|
+
* values fall back to the voice's stored defaults.
|
|
17
|
+
* @see https://elevenlabs.io/docs/api-reference/text-to-speech/convert
|
|
18
|
+
*/
|
|
19
|
+
export interface ElevenLabsVoiceSettings {
|
|
20
|
+
/** Voice stability, 0..1. Default 0.5. */
|
|
21
|
+
stability?: number
|
|
22
|
+
/** Similarity boost, 0..1. Default 0.75. */
|
|
23
|
+
similarityBoost?: number
|
|
24
|
+
/** Style exaggeration, 0..1. Default 0. */
|
|
25
|
+
style?: number
|
|
26
|
+
/** Playback speed. Default 1.0. */
|
|
27
|
+
speed?: number
|
|
28
|
+
/** Clarity/presence boost. Default true. */
|
|
29
|
+
useSpeakerBoost?: boolean
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Provider-specific TTS options. `voice` on `generateSpeech()` takes priority
|
|
34
|
+
* over `voiceId` here, but we expose the same field for callers that prefer
|
|
35
|
+
* to keep voice configuration inside the adapter config.
|
|
36
|
+
*/
|
|
37
|
+
export interface ElevenLabsSpeechProviderOptions {
|
|
38
|
+
/** ElevenLabs voice ID to synthesize. Required if `generateSpeech().voice` is not set. */
|
|
39
|
+
voiceId?: string
|
|
40
|
+
/** Output audio format encoded as `codec_samplerate[_bitrate]`. Defaults to `mp3_44100_128`. */
|
|
41
|
+
outputFormat?: ElevenLabsOutputFormat
|
|
42
|
+
/** Voice-settings overrides for this request only. */
|
|
43
|
+
voiceSettings?: ElevenLabsVoiceSettings
|
|
44
|
+
/** ISO-639-1 language code to enforce (e.g. `'en'`, `'ja'`). */
|
|
45
|
+
languageCode?: string
|
|
46
|
+
/** Deterministic sampling seed, 0..4294967295. */
|
|
47
|
+
seed?: number
|
|
48
|
+
/** Previous text for stitching adjacent clips. */
|
|
49
|
+
previousText?: string
|
|
50
|
+
/** Next text for stitching adjacent clips. */
|
|
51
|
+
nextText?: string
|
|
52
|
+
/** Previous request IDs for stitching (max 3). */
|
|
53
|
+
previousRequestIds?: Array<string>
|
|
54
|
+
/** Next request IDs for stitching (max 3). */
|
|
55
|
+
nextRequestIds?: Array<string>
|
|
56
|
+
/** Text normalization toggle. Default `'auto'`. */
|
|
57
|
+
applyTextNormalization?: 'auto' | 'on' | 'off'
|
|
58
|
+
/** Language-specific text normalization (currently Japanese only, adds latency). */
|
|
59
|
+
applyLanguageTextNormalization?: boolean
|
|
60
|
+
/** Latency optimization level, 0..4. */
|
|
61
|
+
optimizeStreamingLatency?: number
|
|
62
|
+
/** Enable logging. Set false for zero-retention mode (enterprise only). */
|
|
63
|
+
enableLogging?: boolean
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* ElevenLabs text-to-speech adapter built on the official
|
|
68
|
+
* `@elevenlabs/elevenlabs-js` SDK.
|
|
69
|
+
*
|
|
70
|
+
* @example
|
|
71
|
+
* ```ts
|
|
72
|
+
* const adapter = elevenlabsSpeech('eleven_multilingual_v2')
|
|
73
|
+
* const result = await generateSpeech({
|
|
74
|
+
* adapter,
|
|
75
|
+
* text: 'Hello, world!',
|
|
76
|
+
* voice: '21m00Tcm4TlvDq8ikWAM',
|
|
77
|
+
* })
|
|
78
|
+
* ```
|
|
79
|
+
*/
|
|
80
|
+
export class ElevenLabsSpeechAdapter<
|
|
81
|
+
TModel extends ElevenLabsTTSModel,
|
|
82
|
+
> extends BaseTTSAdapter<TModel, ElevenLabsSpeechProviderOptions> {
|
|
83
|
+
readonly name = 'elevenlabs' as const
|
|
84
|
+
|
|
85
|
+
private client: ElevenLabsClient
|
|
86
|
+
|
|
87
|
+
constructor(model: TModel, config?: ElevenLabsClientConfig) {
|
|
88
|
+
super(model, config ?? {})
|
|
89
|
+
this.client = createElevenLabsClient(config)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
async generateSpeech(
|
|
93
|
+
options: TTSOptions<ElevenLabsSpeechProviderOptions>,
|
|
94
|
+
): Promise<TTSResult> {
|
|
95
|
+
const { logger } = options
|
|
96
|
+
logger.request(
|
|
97
|
+
`activity=generateSpeech provider=elevenlabs model=${this.model}`,
|
|
98
|
+
{ provider: 'elevenlabs', model: this.model },
|
|
99
|
+
)
|
|
100
|
+
try {
|
|
101
|
+
const voiceId = options.voice ?? options.modelOptions?.voiceId
|
|
102
|
+
if (!voiceId) {
|
|
103
|
+
throw new Error(
|
|
104
|
+
'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
|
|
105
|
+
)
|
|
106
|
+
}
|
|
107
|
+
const {
|
|
108
|
+
outputFormat,
|
|
109
|
+
voiceSettings,
|
|
110
|
+
languageCode,
|
|
111
|
+
seed,
|
|
112
|
+
previousText,
|
|
113
|
+
nextText,
|
|
114
|
+
previousRequestIds,
|
|
115
|
+
nextRequestIds,
|
|
116
|
+
applyTextNormalization,
|
|
117
|
+
applyLanguageTextNormalization,
|
|
118
|
+
optimizeStreamingLatency,
|
|
119
|
+
enableLogging,
|
|
120
|
+
} = options.modelOptions ?? {}
|
|
121
|
+
const effectiveOutputFormat =
|
|
122
|
+
outputFormat ?? inferOutputFormatFromResponseFormat(options.format)
|
|
123
|
+
|
|
124
|
+
const stream = await this.client.textToSpeech.convert(voiceId, {
|
|
125
|
+
text: options.text,
|
|
126
|
+
modelId: this.model,
|
|
127
|
+
...(effectiveOutputFormat
|
|
128
|
+
? { outputFormat: effectiveOutputFormat }
|
|
129
|
+
: {}),
|
|
130
|
+
...(voiceSettings
|
|
131
|
+
? { voiceSettings: mapVoiceSettings(voiceSettings, options.speed) }
|
|
132
|
+
: options.speed != null
|
|
133
|
+
? { voiceSettings: { speed: options.speed } }
|
|
134
|
+
: {}),
|
|
135
|
+
...(languageCode ? { languageCode } : {}),
|
|
136
|
+
...(seed != null ? { seed } : {}),
|
|
137
|
+
...(previousText ? { previousText } : {}),
|
|
138
|
+
...(nextText ? { nextText } : {}),
|
|
139
|
+
...(previousRequestIds ? { previousRequestIds } : {}),
|
|
140
|
+
...(nextRequestIds ? { nextRequestIds } : {}),
|
|
141
|
+
...(applyTextNormalization ? { applyTextNormalization } : {}),
|
|
142
|
+
...(applyLanguageTextNormalization != null
|
|
143
|
+
? { applyLanguageTextNormalization }
|
|
144
|
+
: {}),
|
|
145
|
+
...(optimizeStreamingLatency != null
|
|
146
|
+
? { optimizeStreamingLatency }
|
|
147
|
+
: {}),
|
|
148
|
+
...(enableLogging != null ? { enableLogging } : {}),
|
|
149
|
+
})
|
|
150
|
+
|
|
151
|
+
const buffer = await readStreamToArrayBuffer(stream)
|
|
152
|
+
const base64 = arrayBufferToBase64(buffer)
|
|
153
|
+
const { format, contentType } = parseOutputFormat(effectiveOutputFormat)
|
|
154
|
+
|
|
155
|
+
return {
|
|
156
|
+
id: generateId(this.name),
|
|
157
|
+
model: this.model,
|
|
158
|
+
audio: base64,
|
|
159
|
+
format,
|
|
160
|
+
contentType,
|
|
161
|
+
}
|
|
162
|
+
} catch (error) {
|
|
163
|
+
logger.errors('elevenlabs.generateSpeech fatal', {
|
|
164
|
+
error,
|
|
165
|
+
source: 'elevenlabs.generateSpeech',
|
|
166
|
+
})
|
|
167
|
+
throw error
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
protected override generateId(): string {
|
|
172
|
+
return generateId(this.name)
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
function mapVoiceSettings(
|
|
177
|
+
settings: ElevenLabsVoiceSettings,
|
|
178
|
+
speedOverride: number | undefined,
|
|
179
|
+
): Record<string, unknown> {
|
|
180
|
+
return {
|
|
181
|
+
...(settings.stability != null ? { stability: settings.stability } : {}),
|
|
182
|
+
...(settings.similarityBoost != null
|
|
183
|
+
? { similarityBoost: settings.similarityBoost }
|
|
184
|
+
: {}),
|
|
185
|
+
...(settings.style != null ? { style: settings.style } : {}),
|
|
186
|
+
...(speedOverride != null
|
|
187
|
+
? { speed: speedOverride }
|
|
188
|
+
: settings.speed != null
|
|
189
|
+
? { speed: settings.speed }
|
|
190
|
+
: {}),
|
|
191
|
+
...(settings.useSpeakerBoost != null
|
|
192
|
+
? { useSpeakerBoost: settings.useSpeakerBoost }
|
|
193
|
+
: {}),
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/**
|
|
198
|
+
* Map the standard TTSOptions `format` (mp3/opus/aac/flac/wav/pcm) to a
|
|
199
|
+
* reasonable ElevenLabs `outputFormat` so callers don't need to know the
|
|
200
|
+
* full codec/samplerate string for the common case.
|
|
201
|
+
*/
|
|
202
|
+
function inferOutputFormatFromResponseFormat(
|
|
203
|
+
format: TTSOptions['format'] | undefined,
|
|
204
|
+
): ElevenLabsOutputFormat | undefined {
|
|
205
|
+
switch (format) {
|
|
206
|
+
case 'mp3':
|
|
207
|
+
return 'mp3_44100_128'
|
|
208
|
+
case 'pcm':
|
|
209
|
+
return 'pcm_44100'
|
|
210
|
+
case 'opus':
|
|
211
|
+
return 'opus_48000_128'
|
|
212
|
+
case undefined:
|
|
213
|
+
return undefined
|
|
214
|
+
default:
|
|
215
|
+
// `aac` / `flac` / `wav` are not native ElevenLabs formats —
|
|
216
|
+
// fall back to mp3 rather than blowing up mid-request.
|
|
217
|
+
return 'mp3_44100_128'
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Create an ElevenLabs speech adapter using `ELEVENLABS_API_KEY` from env.
|
|
223
|
+
*/
|
|
224
|
+
export function elevenlabsSpeech<TModel extends ElevenLabsTTSModel>(
|
|
225
|
+
model: TModel,
|
|
226
|
+
config?: ElevenLabsClientConfig,
|
|
227
|
+
): ElevenLabsSpeechAdapter<TModel> {
|
|
228
|
+
return new ElevenLabsSpeechAdapter(model, config)
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* Create an ElevenLabs speech adapter with an explicit API key.
|
|
233
|
+
*/
|
|
234
|
+
export function createElevenLabsSpeech<TModel extends ElevenLabsTTSModel>(
|
|
235
|
+
model: TModel,
|
|
236
|
+
apiKey: string,
|
|
237
|
+
config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
|
|
238
|
+
): ElevenLabsSpeechAdapter<TModel> {
|
|
239
|
+
return new ElevenLabsSpeechAdapter(model, { apiKey, ...config })
|
|
240
|
+
}
|