@tanstack/ai-elevenlabs 0.4.2 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +2 -2
- package/dist/esm/adapters/audio.js +3 -3
- package/dist/esm/adapters/audio.js.map +1 -1
- package/dist/esm/adapters/speech.d.ts +14 -2
- package/dist/esm/adapters/speech.js +162 -13
- package/dist/esm/adapters/speech.js.map +1 -1
- package/dist/esm/adapters/voice.d.ts +85 -0
- package/dist/esm/adapters/voice.js +154 -0
- package/dist/esm/adapters/voice.js.map +1 -0
- package/dist/esm/index.d.ts +2 -1
- package/dist/esm/index.js +3 -2
- package/dist/esm/model-meta.d.ts +35 -10
- package/dist/esm/model-meta.js +38 -8
- package/dist/esm/model-meta.js.map +1 -1
- package/package.json +5 -5
- package/src/adapters/audio.ts +4 -3
- package/src/adapters/speech.ts +241 -25
- package/src/adapters/voice.ts +299 -0
- package/src/index.ts +13 -0
- package/src/model-meta.ts +61 -13
package/src/adapters/speech.ts
CHANGED
|
@@ -6,8 +6,18 @@ import {
|
|
|
6
6
|
parseOutputFormat,
|
|
7
7
|
readStreamToArrayBuffer,
|
|
8
8
|
} from '../utils/client'
|
|
9
|
-
import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
|
|
10
|
-
import type {
|
|
9
|
+
import type { ElevenLabs, ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
|
|
10
|
+
import type {
|
|
11
|
+
CatalogVoice,
|
|
12
|
+
ListVoicesOptions,
|
|
13
|
+
ListVoicesResult,
|
|
14
|
+
TTSAlignment,
|
|
15
|
+
TTSCapabilities,
|
|
16
|
+
TTSOptions,
|
|
17
|
+
TTSResult,
|
|
18
|
+
TTSSegment,
|
|
19
|
+
VoiceOrigin,
|
|
20
|
+
} from '@tanstack/ai'
|
|
11
21
|
import type { ElevenLabsClientConfig } from '../utils/client'
|
|
12
22
|
import type { ElevenLabsOutputFormat, ElevenLabsTTSModel } from '../model-meta'
|
|
13
23
|
|
|
@@ -37,7 +47,7 @@ export interface ElevenLabsVoiceSettings {
|
|
|
37
47
|
export interface ElevenLabsSpeechProviderOptions {
|
|
38
48
|
/** ElevenLabs voice ID to synthesize. Required if `generateSpeech().voice` is not set. */
|
|
39
49
|
voiceId?: string
|
|
40
|
-
/** Output audio format encoded as `codec_samplerate[_bitrate]`. Defaults to `mp3_44100_128`. */
|
|
50
|
+
/** Output audio format encoded as `codec_samplerate[_bitrate]`. Overrides `format`, including WAV wrapping. Defaults to `mp3_44100_128`. */
|
|
41
51
|
outputFormat?: ElevenLabsOutputFormat
|
|
42
52
|
/** Voice-settings overrides for this request only. */
|
|
43
53
|
voiceSettings?: ElevenLabsVoiceSettings
|
|
@@ -82,6 +92,15 @@ export class ElevenLabsSpeechAdapter<
|
|
|
82
92
|
> extends BaseTTSAdapter<TModel, ElevenLabsSpeechProviderOptions> {
|
|
83
93
|
readonly name = 'elevenlabs' as const
|
|
84
94
|
|
|
95
|
+
/**
|
|
96
|
+
* `textToDialogue` accepts up to 10 distinct voice ids, and both endpoints
|
|
97
|
+
* have a `…WithTimestamps` twin that returns character alignment.
|
|
98
|
+
*/
|
|
99
|
+
override readonly capabilities: TTSCapabilities = {
|
|
100
|
+
maxSpeakers: 10,
|
|
101
|
+
timestamps: true,
|
|
102
|
+
}
|
|
103
|
+
|
|
85
104
|
private readonly client: ElevenLabsClient
|
|
86
105
|
|
|
87
106
|
constructor(model: TModel, config?: ElevenLabsClientConfig) {
|
|
@@ -98,12 +117,6 @@ export class ElevenLabsSpeechAdapter<
|
|
|
98
117
|
{ provider: 'elevenlabs', model: this.model },
|
|
99
118
|
)
|
|
100
119
|
try {
|
|
101
|
-
const voiceId = options.voice ?? options.modelOptions?.voiceId
|
|
102
|
-
if (!voiceId) {
|
|
103
|
-
throw new Error(
|
|
104
|
-
'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
|
|
105
|
-
)
|
|
106
|
-
}
|
|
107
120
|
const {
|
|
108
121
|
outputFormat,
|
|
109
122
|
voiceSettings,
|
|
@@ -120,42 +133,116 @@ export class ElevenLabsSpeechAdapter<
|
|
|
120
133
|
} = options.modelOptions ?? {}
|
|
121
134
|
const effectiveOutputFormat =
|
|
122
135
|
outputFormat ?? inferOutputFormatFromResponseFormat(options.format)
|
|
123
|
-
|
|
124
|
-
const
|
|
125
|
-
|
|
136
|
+
const wrapAsWav = outputFormat == null && options.format === 'wav'
|
|
137
|
+
const { format, contentType } = wrapAsWav
|
|
138
|
+
? { format: 'wav', contentType: 'audio/wav' }
|
|
139
|
+
: parseOutputFormat(effectiveOutputFormat)
|
|
140
|
+
// `format: 'wav'` asks ElevenLabs for pcm_44100 and wraps it here, on
|
|
141
|
+
// every endpoint: the byte-stream pair hands back raw bytes, the
|
|
142
|
+
// timestamped pair hands back the same bytes as base64.
|
|
143
|
+
const encodeAudio = (buffer: ArrayBuffer) =>
|
|
144
|
+
arrayBufferToBase64(wrapAsWav ? wrapPcmAsWav(buffer) : buffer)
|
|
145
|
+
const encodeAudioBase64 = (audioBase64: string) =>
|
|
146
|
+
wrapAsWav ? encodeAudio(base64ToArrayBuffer(audioBase64)) : audioBase64
|
|
147
|
+
const base = {
|
|
126
148
|
modelId: this.model,
|
|
127
149
|
...(effectiveOutputFormat
|
|
128
150
|
? { outputFormat: effectiveOutputFormat }
|
|
129
151
|
: {}),
|
|
152
|
+
...(languageCode ? { languageCode } : {}),
|
|
153
|
+
...(seed != null ? { seed } : {}),
|
|
154
|
+
...(applyTextNormalization ? { applyTextNormalization } : {}),
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
// Four endpoints, picked by (turns?, timestamps?). The dialogue pair
|
|
158
|
+
// takes `inputs` instead of a voice id in the path, and the timestamped
|
|
159
|
+
// pair returns JSON (base64 audio + alignment) instead of a byte stream.
|
|
160
|
+
if (options.turns) {
|
|
161
|
+
const inputs = options.turns.map((turn) => ({
|
|
162
|
+
text: turn.text,
|
|
163
|
+
voiceId: turn.voice,
|
|
164
|
+
}))
|
|
165
|
+
const dialogue = {
|
|
166
|
+
...base,
|
|
167
|
+
inputs,
|
|
168
|
+
...(voiceSettings?.stability != null
|
|
169
|
+
? { settings: { stability: voiceSettings.stability } }
|
|
170
|
+
: {}),
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
if (options.timestamps) {
|
|
174
|
+
const response =
|
|
175
|
+
await this.client.textToDialogue.convertWithTimestamps(dialogue)
|
|
176
|
+
return {
|
|
177
|
+
id: generateId(this.name),
|
|
178
|
+
model: this.model,
|
|
179
|
+
audio: encodeAudioBase64(response.audioBase64),
|
|
180
|
+
format,
|
|
181
|
+
contentType,
|
|
182
|
+
...toAlignmentFields(response.alignment, response.voiceSegments),
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
const stream = await this.client.textToDialogue.convert(dialogue)
|
|
187
|
+
return {
|
|
188
|
+
id: generateId(this.name),
|
|
189
|
+
model: this.model,
|
|
190
|
+
audio: encodeAudio(await readStreamToArrayBuffer(stream)),
|
|
191
|
+
format,
|
|
192
|
+
contentType,
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const voiceId = options.voice ?? options.modelOptions?.voiceId
|
|
197
|
+
if (!voiceId) {
|
|
198
|
+
throw new Error(
|
|
199
|
+
'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
|
|
200
|
+
)
|
|
201
|
+
}
|
|
202
|
+
const single = {
|
|
203
|
+
...base,
|
|
204
|
+
text: options.text,
|
|
130
205
|
...(voiceSettings
|
|
131
206
|
? { voiceSettings: mapVoiceSettings(voiceSettings, options.speed) }
|
|
132
207
|
: options.speed != null
|
|
133
208
|
? { voiceSettings: { speed: options.speed } }
|
|
134
209
|
: {}),
|
|
135
|
-
...(languageCode ? { languageCode } : {}),
|
|
136
|
-
...(seed != null ? { seed } : {}),
|
|
137
210
|
...(previousText ? { previousText } : {}),
|
|
138
211
|
...(nextText ? { nextText } : {}),
|
|
139
212
|
...(previousRequestIds ? { previousRequestIds } : {}),
|
|
140
213
|
...(nextRequestIds ? { nextRequestIds } : {}),
|
|
141
|
-
...(applyTextNormalization ? { applyTextNormalization } : {}),
|
|
142
214
|
...(applyLanguageTextNormalization != null
|
|
143
215
|
? { applyLanguageTextNormalization }
|
|
144
216
|
: {}),
|
|
217
|
+
...(enableLogging != null ? { enableLogging } : {}),
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
if (options.timestamps) {
|
|
221
|
+
const response = await this.client.textToSpeech.convertWithTimestamps(
|
|
222
|
+
voiceId,
|
|
223
|
+
single,
|
|
224
|
+
)
|
|
225
|
+
return {
|
|
226
|
+
id: generateId(this.name),
|
|
227
|
+
model: this.model,
|
|
228
|
+
audio: encodeAudioBase64(response.audioBase64),
|
|
229
|
+
format,
|
|
230
|
+
contentType,
|
|
231
|
+
...toAlignmentFields(response.alignment, undefined),
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
const stream = await this.client.textToSpeech.convert(voiceId, {
|
|
236
|
+
...single,
|
|
145
237
|
...(optimizeStreamingLatency != null
|
|
146
238
|
? { optimizeStreamingLatency }
|
|
147
239
|
: {}),
|
|
148
|
-
...(enableLogging != null ? { enableLogging } : {}),
|
|
149
240
|
})
|
|
150
241
|
|
|
151
|
-
const buffer = await readStreamToArrayBuffer(stream)
|
|
152
|
-
const base64 = arrayBufferToBase64(buffer)
|
|
153
|
-
const { format, contentType } = parseOutputFormat(effectiveOutputFormat)
|
|
154
|
-
|
|
155
242
|
return {
|
|
156
243
|
id: generateId(this.name),
|
|
157
244
|
model: this.model,
|
|
158
|
-
audio:
|
|
245
|
+
audio: encodeAudio(await readStreamToArrayBuffer(stream)),
|
|
159
246
|
format,
|
|
160
247
|
contentType,
|
|
161
248
|
}
|
|
@@ -168,11 +255,73 @@ export class ElevenLabsSpeechAdapter<
|
|
|
168
255
|
}
|
|
169
256
|
}
|
|
170
257
|
|
|
258
|
+
/**
|
|
259
|
+
* List the voices this API key can use, via `GET /v1/voices`.
|
|
260
|
+
*
|
|
261
|
+
* The catalog is per-account and grows every time `generateVoice()` saves a
|
|
262
|
+
* voice, so it has to be read at runtime rather than shipped as a const.
|
|
263
|
+
*/
|
|
264
|
+
override async listVoices(
|
|
265
|
+
options?: ListVoicesOptions,
|
|
266
|
+
): Promise<ListVoicesResult> {
|
|
267
|
+
const response = await this.client.voices.getAll(
|
|
268
|
+
{},
|
|
269
|
+
options?.abortSignal ? { abortSignal: options.abortSignal } : {},
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
const voices = response.voices.map(toCatalogVoice)
|
|
273
|
+
const origins = options?.origins
|
|
274
|
+
// ElevenLabs has no origin filter on /v1/voices, so narrow in memory.
|
|
275
|
+
return {
|
|
276
|
+
voices: origins
|
|
277
|
+
? voices.filter(
|
|
278
|
+
(voice) => voice.origin && origins.includes(voice.origin),
|
|
279
|
+
)
|
|
280
|
+
: voices,
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
|
|
171
284
|
protected override generateId(): string {
|
|
172
285
|
return generateId(this.name)
|
|
173
286
|
}
|
|
174
287
|
}
|
|
175
288
|
|
|
289
|
+
/**
|
|
290
|
+
* Map the ElevenLabs timestamp payload onto the core `alignment` / `segments`
|
|
291
|
+
* fields. `voiceSegments` only comes back from the dialogue endpoint; its
|
|
292
|
+
* character indices slice the alignment into per-turn text.
|
|
293
|
+
*/
|
|
294
|
+
function toAlignmentFields(
|
|
295
|
+
alignment: ElevenLabs.CharacterAlignmentResponseModel | undefined,
|
|
296
|
+
voiceSegments: Array<ElevenLabs.VoiceSegment> | undefined,
|
|
297
|
+
): { alignment?: TTSAlignment; segments?: Array<TTSSegment> } {
|
|
298
|
+
const fields: { alignment?: TTSAlignment; segments?: Array<TTSSegment> } = {}
|
|
299
|
+
if (alignment) {
|
|
300
|
+
fields.alignment = {
|
|
301
|
+
unit: 'character',
|
|
302
|
+
texts: alignment.characters,
|
|
303
|
+
startSeconds: alignment.characterStartTimesSeconds,
|
|
304
|
+
endSeconds: alignment.characterEndTimesSeconds,
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
if (voiceSegments && voiceSegments.length > 0) {
|
|
308
|
+
fields.segments = voiceSegments.map((segment) => ({
|
|
309
|
+
startSeconds: segment.startTimeSeconds,
|
|
310
|
+
endSeconds: segment.endTimeSeconds,
|
|
311
|
+
turnIndex: segment.dialogueInputIndex,
|
|
312
|
+
voice: segment.voiceId,
|
|
313
|
+
...(alignment
|
|
314
|
+
? {
|
|
315
|
+
text: alignment.characters
|
|
316
|
+
.slice(segment.characterStartIndex, segment.characterEndIndex)
|
|
317
|
+
.join(''),
|
|
318
|
+
}
|
|
319
|
+
: {}),
|
|
320
|
+
}))
|
|
321
|
+
}
|
|
322
|
+
return fields
|
|
323
|
+
}
|
|
324
|
+
|
|
176
325
|
function mapVoiceSettings(
|
|
177
326
|
settings: ElevenLabsVoiceSettings,
|
|
178
327
|
speedOverride: number | undefined,
|
|
@@ -194,6 +343,44 @@ function mapVoiceSettings(
|
|
|
194
343
|
}
|
|
195
344
|
}
|
|
196
345
|
|
|
346
|
+
/**
|
|
347
|
+
* ElevenLabs voice categories map onto {@link VoiceOrigin} with two joins:
|
|
348
|
+
* `famous` and `high_quality` are both curated tiers, so both read as
|
|
349
|
+
* `'professional'`. An unrecognized category is dropped rather than guessed,
|
|
350
|
+
* which keeps an `origins` filter from silently matching the wrong thing.
|
|
351
|
+
*/
|
|
352
|
+
function toVoiceOrigin(
|
|
353
|
+
category: ElevenLabs.VoiceCategory | undefined,
|
|
354
|
+
): VoiceOrigin | undefined {
|
|
355
|
+
switch (category) {
|
|
356
|
+
case 'premade':
|
|
357
|
+
return 'premade'
|
|
358
|
+
case 'generated':
|
|
359
|
+
return 'generated'
|
|
360
|
+
case 'cloned':
|
|
361
|
+
return 'cloned'
|
|
362
|
+
case 'professional':
|
|
363
|
+
case 'famous':
|
|
364
|
+
case 'high_quality':
|
|
365
|
+
return 'professional'
|
|
366
|
+
case undefined:
|
|
367
|
+
default:
|
|
368
|
+
return undefined
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
function toCatalogVoice(voice: ElevenLabs.Voice): CatalogVoice {
|
|
373
|
+
const origin = toVoiceOrigin(voice.category)
|
|
374
|
+
return {
|
|
375
|
+
voiceId: voice.voiceId,
|
|
376
|
+
...(voice.name ? { name: voice.name } : {}),
|
|
377
|
+
...(origin ? { origin } : {}),
|
|
378
|
+
...(voice.description ? { description: voice.description } : {}),
|
|
379
|
+
...(voice.previewUrl ? { previewUrl: voice.previewUrl } : {}),
|
|
380
|
+
...(voice.labels ? { labels: voice.labels } : {}),
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
|
|
197
384
|
/**
|
|
198
385
|
* Map the standard TTSOptions `format` (mp3/opus/aac/flac/wav/pcm) to a
|
|
199
386
|
* reasonable ElevenLabs `outputFormat` so callers don't need to know the
|
|
@@ -206,6 +393,7 @@ function inferOutputFormatFromResponseFormat(
|
|
|
206
393
|
case 'mp3':
|
|
207
394
|
return 'mp3_44100_128'
|
|
208
395
|
case 'pcm':
|
|
396
|
+
case 'wav':
|
|
209
397
|
return 'pcm_44100'
|
|
210
398
|
case 'opus':
|
|
211
399
|
return 'opus_48000_128'
|
|
@@ -213,14 +401,42 @@ function inferOutputFormatFromResponseFormat(
|
|
|
213
401
|
return undefined
|
|
214
402
|
case 'aac':
|
|
215
403
|
case 'flac':
|
|
216
|
-
case 'wav':
|
|
217
404
|
default:
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
405
|
+
throw new Error(
|
|
406
|
+
`ElevenLabs TTS does not support format '${format}'. Use mp3, pcm, opus, or wav.`,
|
|
407
|
+
)
|
|
221
408
|
}
|
|
222
409
|
}
|
|
223
410
|
|
|
411
|
+
function base64ToArrayBuffer(base64: string): ArrayBuffer {
|
|
412
|
+
const binary = atob(base64)
|
|
413
|
+
const bytes = new Uint8Array(binary.length)
|
|
414
|
+
for (let i = 0; i < binary.length; i += 1) bytes[i] = binary.charCodeAt(i)
|
|
415
|
+
return bytes.buffer
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
/** Wrap ElevenLabs pcm_44100 (16-bit little-endian mono) in a RIFF/WAV container. */
|
|
419
|
+
function wrapPcmAsWav(pcm: ArrayBuffer): ArrayBuffer {
|
|
420
|
+
const wav = new ArrayBuffer(44 + pcm.byteLength)
|
|
421
|
+
const bytes = new Uint8Array(wav)
|
|
422
|
+
const header = new DataView(wav)
|
|
423
|
+
const encoder = new TextEncoder()
|
|
424
|
+
bytes.set(encoder.encode('RIFF'), 0)
|
|
425
|
+
header.setUint32(4, 36 + pcm.byteLength, true)
|
|
426
|
+
bytes.set(encoder.encode('WAVEfmt '), 8)
|
|
427
|
+
header.setUint32(16, 16, true) // PCM format chunk size
|
|
428
|
+
header.setUint16(20, 1, true) // PCM encoding
|
|
429
|
+
header.setUint16(22, 1, true) // mono
|
|
430
|
+
header.setUint32(24, 44100, true) // sample rate
|
|
431
|
+
header.setUint32(28, 44100 * 2, true) // bytes per second
|
|
432
|
+
header.setUint16(32, 2, true) // bytes per sample
|
|
433
|
+
header.setUint16(34, 16, true) // bits per sample
|
|
434
|
+
bytes.set(encoder.encode('data'), 36)
|
|
435
|
+
header.setUint32(40, pcm.byteLength, true)
|
|
436
|
+
bytes.set(new Uint8Array(pcm), 44)
|
|
437
|
+
return wav
|
|
438
|
+
}
|
|
439
|
+
|
|
224
440
|
/**
|
|
225
441
|
* Create an ElevenLabs speech adapter using `ELEVENLABS_API_KEY` from env.
|
|
226
442
|
*/
|
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
import { BaseVoiceAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import {
|
|
3
|
+
arrayBufferToBase64,
|
|
4
|
+
createElevenLabsClient,
|
|
5
|
+
generateId,
|
|
6
|
+
parseOutputFormat,
|
|
7
|
+
} from '../utils/client'
|
|
8
|
+
import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
|
|
9
|
+
import type {
|
|
10
|
+
GeneratedVoice,
|
|
11
|
+
VoiceGenerationOptions,
|
|
12
|
+
VoiceResult,
|
|
13
|
+
} from '@tanstack/ai'
|
|
14
|
+
import type { ElevenLabsClientConfig } from '../utils/client'
|
|
15
|
+
import type {
|
|
16
|
+
ElevenLabsOutputFormat,
|
|
17
|
+
ElevenLabsVoiceModel,
|
|
18
|
+
} from '../model-meta'
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Provider-specific voice-design options. Fields map 1:1 onto the SDK's
|
|
22
|
+
* `VoiceDesignRequestModel` — mirroring the names so ElevenLabs'
|
|
23
|
+
* documentation stays useful here.
|
|
24
|
+
* @see https://elevenlabs.io/docs/api-reference/text-to-voice/design
|
|
25
|
+
*/
|
|
26
|
+
export interface ElevenLabsVoiceProviderOptions {
|
|
27
|
+
/** Output audio format for the previews, `codec_samplerate[_bitrate]`. */
|
|
28
|
+
outputFormat?: ElevenLabsOutputFormat
|
|
29
|
+
/** Line the previews speak, 100..1000 characters. */
|
|
30
|
+
text?: string
|
|
31
|
+
/** Let ElevenLabs write the preview line from the description. */
|
|
32
|
+
autoGenerateText?: boolean
|
|
33
|
+
/** Preview loudness, -1 (quietest) to 1 (loudest). 0 is roughly -24 LUFS. */
|
|
34
|
+
loudness?: number
|
|
35
|
+
/** Deterministic sampling seed — same seed and inputs produce the same voice. */
|
|
36
|
+
seed?: number
|
|
37
|
+
/** How closely to follow the description. High values can sound robotic. */
|
|
38
|
+
guidanceScale?: number
|
|
39
|
+
/** Higher quality trades variety for fidelity. */
|
|
40
|
+
quality?: number
|
|
41
|
+
/** Let ElevenLabs expand a short description into a detailed one. */
|
|
42
|
+
shouldEnhance?: boolean
|
|
43
|
+
/**
|
|
44
|
+
* Balance of description against reference audio, 0 (almost all reference)
|
|
45
|
+
* to 1 (almost all description). `eleven_ttv_v3` only.
|
|
46
|
+
*/
|
|
47
|
+
promptStrength?: number
|
|
48
|
+
/** Metadata stored on the voice. Only used when the voice is saved. */
|
|
49
|
+
labels?: Record<string, string>
|
|
50
|
+
/** Remixing session to attach these generations to. */
|
|
51
|
+
remixingSessionId?: string
|
|
52
|
+
/** Remixing session iteration to attach these generations to. */
|
|
53
|
+
remixingSessionIterationId?: string
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* ElevenLabs voice-design adapter built on the official
|
|
58
|
+
* `@elevenlabs/elevenlabs-js` SDK.
|
|
59
|
+
*
|
|
60
|
+
* ElevenLabs designs voices in two steps — generate previews, then promote one
|
|
61
|
+
* into a real voice — but that is an implementation detail. Pass `name` to get
|
|
62
|
+
* a saved voice back; leave it off to audition previews first.
|
|
63
|
+
*
|
|
64
|
+
* @example Audition previews
|
|
65
|
+
* ```ts
|
|
66
|
+
* const result = await generateVoice({
|
|
67
|
+
* adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
|
|
68
|
+
* prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
|
|
69
|
+
* })
|
|
70
|
+
* ```
|
|
71
|
+
*
|
|
72
|
+
* @example Save the voice
|
|
73
|
+
* ```ts
|
|
74
|
+
* const result = await generateVoice({
|
|
75
|
+
* adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
|
|
76
|
+
* prompt: 'A bright, upbeat product demo host',
|
|
77
|
+
* name: 'Demo Host',
|
|
78
|
+
* })
|
|
79
|
+
* ```
|
|
80
|
+
*/
|
|
81
|
+
export class ElevenLabsVoiceAdapter<
|
|
82
|
+
TModel extends ElevenLabsVoiceModel,
|
|
83
|
+
> extends BaseVoiceAdapter<TModel, ElevenLabsVoiceProviderOptions> {
|
|
84
|
+
readonly name = 'elevenlabs' as const
|
|
85
|
+
|
|
86
|
+
private readonly client: ElevenLabsClient
|
|
87
|
+
|
|
88
|
+
constructor(model: TModel, config?: ElevenLabsClientConfig) {
|
|
89
|
+
super(model, config ?? {})
|
|
90
|
+
this.client = createElevenLabsClient(config)
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
async generateVoice(
|
|
94
|
+
options: VoiceGenerationOptions<ElevenLabsVoiceProviderOptions>,
|
|
95
|
+
): Promise<VoiceResult> {
|
|
96
|
+
const { logger } = options
|
|
97
|
+
logger.request(
|
|
98
|
+
`activity=generateVoice provider=elevenlabs model=${this.model}`,
|
|
99
|
+
{ provider: 'elevenlabs', model: this.model },
|
|
100
|
+
)
|
|
101
|
+
try {
|
|
102
|
+
// `voiceDescription` is required by the design endpoint, so reference
|
|
103
|
+
// audio alone is not enough here — unlike clone-only providers.
|
|
104
|
+
if (!options.prompt) {
|
|
105
|
+
throw new Error(
|
|
106
|
+
'ElevenLabs voice design requires a `prompt` describing the voice. Reference audio alone is not supported; pass both to guide the design with a real speaker.',
|
|
107
|
+
)
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
const opts = options.modelOptions ?? {}
|
|
111
|
+
const referenceAudioBase64 = await toReferenceAudioBase64(
|
|
112
|
+
options.referenceAudio,
|
|
113
|
+
)
|
|
114
|
+
if (referenceAudioBase64 && this.model !== 'eleven_ttv_v3') {
|
|
115
|
+
throw new Error(
|
|
116
|
+
`ElevenLabs only accepts reference audio on eleven_ttv_v3, but this adapter is using "${this.model}".`,
|
|
117
|
+
)
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const requestOptions = options.abortSignal
|
|
121
|
+
? { abortSignal: options.abortSignal }
|
|
122
|
+
: {}
|
|
123
|
+
|
|
124
|
+
const previewResponse = await this.client.textToVoice.design(
|
|
125
|
+
{
|
|
126
|
+
voiceDescription: options.prompt,
|
|
127
|
+
modelId: this.model,
|
|
128
|
+
...(opts.outputFormat ? { outputFormat: opts.outputFormat } : {}),
|
|
129
|
+
...(opts.text ? { text: opts.text } : {}),
|
|
130
|
+
...(opts.autoGenerateText != null
|
|
131
|
+
? { autoGenerateText: opts.autoGenerateText }
|
|
132
|
+
: {}),
|
|
133
|
+
...(opts.loudness != null ? { loudness: opts.loudness } : {}),
|
|
134
|
+
...(opts.seed != null ? { seed: opts.seed } : {}),
|
|
135
|
+
...(opts.guidanceScale != null
|
|
136
|
+
? { guidanceScale: opts.guidanceScale }
|
|
137
|
+
: {}),
|
|
138
|
+
...(opts.quality != null ? { quality: opts.quality } : {}),
|
|
139
|
+
...(opts.shouldEnhance != null
|
|
140
|
+
? { shouldEnhance: opts.shouldEnhance }
|
|
141
|
+
: {}),
|
|
142
|
+
...(opts.promptStrength != null
|
|
143
|
+
? { promptStrength: opts.promptStrength }
|
|
144
|
+
: {}),
|
|
145
|
+
...(referenceAudioBase64 ? { referenceAudioBase64 } : {}),
|
|
146
|
+
...(opts.remixingSessionId
|
|
147
|
+
? { remixingSessionId: opts.remixingSessionId }
|
|
148
|
+
: {}),
|
|
149
|
+
...(opts.remixingSessionIterationId
|
|
150
|
+
? { remixingSessionIterationId: opts.remixingSessionIterationId }
|
|
151
|
+
: {}),
|
|
152
|
+
},
|
|
153
|
+
requestOptions,
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
const { format, contentType } = parseOutputFormat(opts.outputFormat)
|
|
157
|
+
const voices: Array<GeneratedVoice> = previewResponse.previews.map(
|
|
158
|
+
(preview) => ({
|
|
159
|
+
voiceId: preview.generatedVoiceId,
|
|
160
|
+
audio: preview.audioBase64,
|
|
161
|
+
format,
|
|
162
|
+
contentType: preview.mediaType || contentType,
|
|
163
|
+
duration: preview.durationSecs,
|
|
164
|
+
...(preview.language ? { language: preview.language } : {}),
|
|
165
|
+
saved: false,
|
|
166
|
+
// The design endpoint returns a finished preview; nothing trains.
|
|
167
|
+
status: 'ready' as const,
|
|
168
|
+
}),
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
// A caller who passed `name` asked for a persisted voice. Returning an
|
|
172
|
+
// empty list would report success for a request that produced nothing.
|
|
173
|
+
if (voices.length === 0) {
|
|
174
|
+
throw new Error(
|
|
175
|
+
'ElevenLabs returned no voice previews for this description. Try a longer, more specific prompt.',
|
|
176
|
+
)
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
if (options.name) {
|
|
180
|
+
voices[0] = await this.promoteFirstPreview(
|
|
181
|
+
voices,
|
|
182
|
+
options.name,
|
|
183
|
+
options.description ?? options.prompt,
|
|
184
|
+
opts.labels,
|
|
185
|
+
requestOptions,
|
|
186
|
+
)
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
return {
|
|
190
|
+
id: generateId(this.name),
|
|
191
|
+
model: this.model,
|
|
192
|
+
voices,
|
|
193
|
+
previewText: previewResponse.text,
|
|
194
|
+
}
|
|
195
|
+
} catch (error) {
|
|
196
|
+
logger.errors('elevenlabs.generateVoice fatal', {
|
|
197
|
+
error,
|
|
198
|
+
source: 'elevenlabs.generateVoice',
|
|
199
|
+
})
|
|
200
|
+
throw error
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Promote the best preview into a real library voice. The remaining
|
|
206
|
+
* generated ids go along as `playedNotSelectedVoiceIds` — ElevenLabs uses
|
|
207
|
+
* them as RLHF signal for future designs.
|
|
208
|
+
*/
|
|
209
|
+
private async promoteFirstPreview(
|
|
210
|
+
voices: ReadonlyArray<GeneratedVoice>,
|
|
211
|
+
voiceName: string,
|
|
212
|
+
voiceDescription: string,
|
|
213
|
+
labels: Record<string, string> | undefined,
|
|
214
|
+
requestOptions: { abortSignal?: AbortSignal },
|
|
215
|
+
): Promise<GeneratedVoice> {
|
|
216
|
+
const [best, ...rest] = voices
|
|
217
|
+
if (!best) {
|
|
218
|
+
throw new Error('No preview to promote into a library voice.')
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
const saved = await this.client.textToVoice.create(
|
|
222
|
+
{
|
|
223
|
+
voiceName,
|
|
224
|
+
voiceDescription,
|
|
225
|
+
generatedVoiceId: best.voiceId,
|
|
226
|
+
...(labels ? { labels } : {}),
|
|
227
|
+
...(rest.length > 0
|
|
228
|
+
? { playedNotSelectedVoiceIds: rest.map((voice) => voice.voiceId) }
|
|
229
|
+
: {}),
|
|
230
|
+
},
|
|
231
|
+
requestOptions,
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
return { ...best, voiceId: saved.voiceId, saved: true }
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
protected override generateId(): string {
|
|
238
|
+
return generateId(this.name)
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/**
|
|
243
|
+
* Normalize reference audio to the bare base64 the design endpoint wants.
|
|
244
|
+
*
|
|
245
|
+
* https URLs are rejected rather than fetched: ElevenLabs has no URL field
|
|
246
|
+
* here, and downloading caller-supplied media into memory to inline it is a
|
|
247
|
+
* footgun on large files.
|
|
248
|
+
*/
|
|
249
|
+
async function toReferenceAudioBase64(
|
|
250
|
+
audio: VoiceGenerationOptions['referenceAudio'],
|
|
251
|
+
): Promise<string | undefined> {
|
|
252
|
+
if (audio == null) return undefined
|
|
253
|
+
if (audio instanceof ArrayBuffer) return arrayBufferToBase64(audio)
|
|
254
|
+
if (typeof audio !== 'string') {
|
|
255
|
+
return arrayBufferToBase64(await audio.arrayBuffer())
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
if (audio.startsWith('data:')) {
|
|
259
|
+
const commaIndex = audio.indexOf(',')
|
|
260
|
+
const header = commaIndex === -1 ? '' : audio.slice(5, commaIndex)
|
|
261
|
+
if (commaIndex === -1 || !/;base64$/i.test(header)) {
|
|
262
|
+
throw new Error(
|
|
263
|
+
'ElevenLabs voice design needs base64 reference audio. Pass a base64 data URL, a base64 string, a Blob, or an ArrayBuffer.',
|
|
264
|
+
)
|
|
265
|
+
}
|
|
266
|
+
return audio.slice(commaIndex + 1)
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
if (/^https?:\/\//i.test(audio)) {
|
|
270
|
+
throw new Error(
|
|
271
|
+
'ElevenLabs voice design does not accept reference audio URLs. Read the file yourself and pass a Blob, ArrayBuffer, or base64 string.',
|
|
272
|
+
)
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
return audio
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* Create an ElevenLabs voice-design adapter using `ELEVENLABS_API_KEY` from env.
|
|
280
|
+
*/
|
|
281
|
+
export function elevenlabsVoiceDesign<TModel extends ElevenLabsVoiceModel>(
|
|
282
|
+
model: TModel,
|
|
283
|
+
config?: ElevenLabsClientConfig,
|
|
284
|
+
): ElevenLabsVoiceAdapter<TModel> {
|
|
285
|
+
return new ElevenLabsVoiceAdapter(model, config)
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* Create an ElevenLabs voice-design adapter with an explicit API key.
|
|
290
|
+
*/
|
|
291
|
+
export function createElevenLabsVoiceDesign<
|
|
292
|
+
TModel extends ElevenLabsVoiceModel,
|
|
293
|
+
>(
|
|
294
|
+
model: TModel,
|
|
295
|
+
apiKey: string,
|
|
296
|
+
config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
|
|
297
|
+
): ElevenLabsVoiceAdapter<TModel> {
|
|
298
|
+
return new ElevenLabsVoiceAdapter(model, { apiKey, ...config })
|
|
299
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -38,6 +38,17 @@ export {
|
|
|
38
38
|
type ElevenLabsMusicCompositionPlan,
|
|
39
39
|
} from './adapters/audio'
|
|
40
40
|
|
|
41
|
+
// ============================================================================
|
|
42
|
+
// Voice (Voice Design) Adapter
|
|
43
|
+
// ============================================================================
|
|
44
|
+
|
|
45
|
+
export {
|
|
46
|
+
ElevenLabsVoiceAdapter,
|
|
47
|
+
createElevenLabsVoiceDesign,
|
|
48
|
+
elevenlabsVoiceDesign,
|
|
49
|
+
type ElevenLabsVoiceProviderOptions,
|
|
50
|
+
} from './adapters/voice'
|
|
51
|
+
|
|
41
52
|
// ============================================================================
|
|
42
53
|
// Transcription (Speech-to-Text) Adapter
|
|
43
54
|
// ============================================================================
|
|
@@ -57,6 +68,7 @@ export {
|
|
|
57
68
|
ELEVENLABS_TTS_MODELS,
|
|
58
69
|
ELEVENLABS_AUDIO_MODELS,
|
|
59
70
|
ELEVENLABS_TRANSCRIPTION_MODELS,
|
|
71
|
+
ELEVENLABS_VOICE_MODELS,
|
|
60
72
|
isElevenLabsMusicModel,
|
|
61
73
|
isElevenLabsSoundEffectsModel,
|
|
62
74
|
type ElevenLabsTTSModel,
|
|
@@ -64,6 +76,7 @@ export {
|
|
|
64
76
|
type ElevenLabsMusicModel,
|
|
65
77
|
type ElevenLabsSoundEffectsModel,
|
|
66
78
|
type ElevenLabsTranscriptionModel,
|
|
79
|
+
type ElevenLabsVoiceModel,
|
|
67
80
|
type ElevenLabsOutputFormat,
|
|
68
81
|
} from './model-meta'
|
|
69
82
|
|