@tanstack/ai-elevenlabs 0.4.2 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,8 +6,18 @@ import {
6
6
  parseOutputFormat,
7
7
  readStreamToArrayBuffer,
8
8
  } from '../utils/client'
9
- import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
10
- import type { TTSOptions, TTSResult } from '@tanstack/ai'
9
+ import type { ElevenLabs, ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
10
+ import type {
11
+ CatalogVoice,
12
+ ListVoicesOptions,
13
+ ListVoicesResult,
14
+ TTSAlignment,
15
+ TTSCapabilities,
16
+ TTSOptions,
17
+ TTSResult,
18
+ TTSSegment,
19
+ VoiceOrigin,
20
+ } from '@tanstack/ai'
11
21
  import type { ElevenLabsClientConfig } from '../utils/client'
12
22
  import type { ElevenLabsOutputFormat, ElevenLabsTTSModel } from '../model-meta'
13
23
 
@@ -37,7 +47,7 @@ export interface ElevenLabsVoiceSettings {
37
47
  export interface ElevenLabsSpeechProviderOptions {
38
48
  /** ElevenLabs voice ID to synthesize. Required if `generateSpeech().voice` is not set. */
39
49
  voiceId?: string
40
- /** Output audio format encoded as `codec_samplerate[_bitrate]`. Defaults to `mp3_44100_128`. */
50
+ /** Output audio format encoded as `codec_samplerate[_bitrate]`. Overrides `format`, including WAV wrapping. Defaults to `mp3_44100_128`. */
41
51
  outputFormat?: ElevenLabsOutputFormat
42
52
  /** Voice-settings overrides for this request only. */
43
53
  voiceSettings?: ElevenLabsVoiceSettings
@@ -82,6 +92,15 @@ export class ElevenLabsSpeechAdapter<
82
92
  > extends BaseTTSAdapter<TModel, ElevenLabsSpeechProviderOptions> {
83
93
  readonly name = 'elevenlabs' as const
84
94
 
95
+ /**
96
+ * `textToDialogue` accepts up to 10 distinct voice ids, and both endpoints
97
+ * have a `…WithTimestamps` twin that returns character alignment.
98
+ */
99
+ override readonly capabilities: TTSCapabilities = {
100
+ maxSpeakers: 10,
101
+ timestamps: true,
102
+ }
103
+
85
104
  private readonly client: ElevenLabsClient
86
105
 
87
106
  constructor(model: TModel, config?: ElevenLabsClientConfig) {
@@ -98,12 +117,6 @@ export class ElevenLabsSpeechAdapter<
98
117
  { provider: 'elevenlabs', model: this.model },
99
118
  )
100
119
  try {
101
- const voiceId = options.voice ?? options.modelOptions?.voiceId
102
- if (!voiceId) {
103
- throw new Error(
104
- 'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
105
- )
106
- }
107
120
  const {
108
121
  outputFormat,
109
122
  voiceSettings,
@@ -120,42 +133,116 @@ export class ElevenLabsSpeechAdapter<
120
133
  } = options.modelOptions ?? {}
121
134
  const effectiveOutputFormat =
122
135
  outputFormat ?? inferOutputFormatFromResponseFormat(options.format)
123
-
124
- const stream = await this.client.textToSpeech.convert(voiceId, {
125
- text: options.text,
136
+ const wrapAsWav = outputFormat == null && options.format === 'wav'
137
+ const { format, contentType } = wrapAsWav
138
+ ? { format: 'wav', contentType: 'audio/wav' }
139
+ : parseOutputFormat(effectiveOutputFormat)
140
+ // `format: 'wav'` asks ElevenLabs for pcm_44100 and wraps it here, on
141
+ // every endpoint: the byte-stream pair hands back raw bytes, the
142
+ // timestamped pair hands back the same bytes as base64.
143
+ const encodeAudio = (buffer: ArrayBuffer) =>
144
+ arrayBufferToBase64(wrapAsWav ? wrapPcmAsWav(buffer) : buffer)
145
+ const encodeAudioBase64 = (audioBase64: string) =>
146
+ wrapAsWav ? encodeAudio(base64ToArrayBuffer(audioBase64)) : audioBase64
147
+ const base = {
126
148
  modelId: this.model,
127
149
  ...(effectiveOutputFormat
128
150
  ? { outputFormat: effectiveOutputFormat }
129
151
  : {}),
152
+ ...(languageCode ? { languageCode } : {}),
153
+ ...(seed != null ? { seed } : {}),
154
+ ...(applyTextNormalization ? { applyTextNormalization } : {}),
155
+ }
156
+
157
+ // Four endpoints, picked by (turns?, timestamps?). The dialogue pair
158
+ // takes `inputs` instead of a voice id in the path, and the timestamped
159
+ // pair returns JSON (base64 audio + alignment) instead of a byte stream.
160
+ if (options.turns) {
161
+ const inputs = options.turns.map((turn) => ({
162
+ text: turn.text,
163
+ voiceId: turn.voice,
164
+ }))
165
+ const dialogue = {
166
+ ...base,
167
+ inputs,
168
+ ...(voiceSettings?.stability != null
169
+ ? { settings: { stability: voiceSettings.stability } }
170
+ : {}),
171
+ }
172
+
173
+ if (options.timestamps) {
174
+ const response =
175
+ await this.client.textToDialogue.convertWithTimestamps(dialogue)
176
+ return {
177
+ id: generateId(this.name),
178
+ model: this.model,
179
+ audio: encodeAudioBase64(response.audioBase64),
180
+ format,
181
+ contentType,
182
+ ...toAlignmentFields(response.alignment, response.voiceSegments),
183
+ }
184
+ }
185
+
186
+ const stream = await this.client.textToDialogue.convert(dialogue)
187
+ return {
188
+ id: generateId(this.name),
189
+ model: this.model,
190
+ audio: encodeAudio(await readStreamToArrayBuffer(stream)),
191
+ format,
192
+ contentType,
193
+ }
194
+ }
195
+
196
+ const voiceId = options.voice ?? options.modelOptions?.voiceId
197
+ if (!voiceId) {
198
+ throw new Error(
199
+ 'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
200
+ )
201
+ }
202
+ const single = {
203
+ ...base,
204
+ text: options.text,
130
205
  ...(voiceSettings
131
206
  ? { voiceSettings: mapVoiceSettings(voiceSettings, options.speed) }
132
207
  : options.speed != null
133
208
  ? { voiceSettings: { speed: options.speed } }
134
209
  : {}),
135
- ...(languageCode ? { languageCode } : {}),
136
- ...(seed != null ? { seed } : {}),
137
210
  ...(previousText ? { previousText } : {}),
138
211
  ...(nextText ? { nextText } : {}),
139
212
  ...(previousRequestIds ? { previousRequestIds } : {}),
140
213
  ...(nextRequestIds ? { nextRequestIds } : {}),
141
- ...(applyTextNormalization ? { applyTextNormalization } : {}),
142
214
  ...(applyLanguageTextNormalization != null
143
215
  ? { applyLanguageTextNormalization }
144
216
  : {}),
217
+ ...(enableLogging != null ? { enableLogging } : {}),
218
+ }
219
+
220
+ if (options.timestamps) {
221
+ const response = await this.client.textToSpeech.convertWithTimestamps(
222
+ voiceId,
223
+ single,
224
+ )
225
+ return {
226
+ id: generateId(this.name),
227
+ model: this.model,
228
+ audio: encodeAudioBase64(response.audioBase64),
229
+ format,
230
+ contentType,
231
+ ...toAlignmentFields(response.alignment, undefined),
232
+ }
233
+ }
234
+
235
+ const stream = await this.client.textToSpeech.convert(voiceId, {
236
+ ...single,
145
237
  ...(optimizeStreamingLatency != null
146
238
  ? { optimizeStreamingLatency }
147
239
  : {}),
148
- ...(enableLogging != null ? { enableLogging } : {}),
149
240
  })
150
241
 
151
- const buffer = await readStreamToArrayBuffer(stream)
152
- const base64 = arrayBufferToBase64(buffer)
153
- const { format, contentType } = parseOutputFormat(effectiveOutputFormat)
154
-
155
242
  return {
156
243
  id: generateId(this.name),
157
244
  model: this.model,
158
- audio: base64,
245
+ audio: encodeAudio(await readStreamToArrayBuffer(stream)),
159
246
  format,
160
247
  contentType,
161
248
  }
@@ -168,11 +255,73 @@ export class ElevenLabsSpeechAdapter<
168
255
  }
169
256
  }
170
257
 
258
+ /**
259
+ * List the voices this API key can use, via `GET /v1/voices`.
260
+ *
261
+ * The catalog is per-account and grows every time `generateVoice()` saves a
262
+ * voice, so it has to be read at runtime rather than shipped as a const.
263
+ */
264
+ override async listVoices(
265
+ options?: ListVoicesOptions,
266
+ ): Promise<ListVoicesResult> {
267
+ const response = await this.client.voices.getAll(
268
+ {},
269
+ options?.abortSignal ? { abortSignal: options.abortSignal } : {},
270
+ )
271
+
272
+ const voices = response.voices.map(toCatalogVoice)
273
+ const origins = options?.origins
274
+ // ElevenLabs has no origin filter on /v1/voices, so narrow in memory.
275
+ return {
276
+ voices: origins
277
+ ? voices.filter(
278
+ (voice) => voice.origin && origins.includes(voice.origin),
279
+ )
280
+ : voices,
281
+ }
282
+ }
283
+
171
284
  protected override generateId(): string {
172
285
  return generateId(this.name)
173
286
  }
174
287
  }
175
288
 
289
+ /**
290
+ * Map the ElevenLabs timestamp payload onto the core `alignment` / `segments`
291
+ * fields. `voiceSegments` only comes back from the dialogue endpoint; its
292
+ * character indices slice the alignment into per-turn text.
293
+ */
294
+ function toAlignmentFields(
295
+ alignment: ElevenLabs.CharacterAlignmentResponseModel | undefined,
296
+ voiceSegments: Array<ElevenLabs.VoiceSegment> | undefined,
297
+ ): { alignment?: TTSAlignment; segments?: Array<TTSSegment> } {
298
+ const fields: { alignment?: TTSAlignment; segments?: Array<TTSSegment> } = {}
299
+ if (alignment) {
300
+ fields.alignment = {
301
+ unit: 'character',
302
+ texts: alignment.characters,
303
+ startSeconds: alignment.characterStartTimesSeconds,
304
+ endSeconds: alignment.characterEndTimesSeconds,
305
+ }
306
+ }
307
+ if (voiceSegments && voiceSegments.length > 0) {
308
+ fields.segments = voiceSegments.map((segment) => ({
309
+ startSeconds: segment.startTimeSeconds,
310
+ endSeconds: segment.endTimeSeconds,
311
+ turnIndex: segment.dialogueInputIndex,
312
+ voice: segment.voiceId,
313
+ ...(alignment
314
+ ? {
315
+ text: alignment.characters
316
+ .slice(segment.characterStartIndex, segment.characterEndIndex)
317
+ .join(''),
318
+ }
319
+ : {}),
320
+ }))
321
+ }
322
+ return fields
323
+ }
324
+
176
325
  function mapVoiceSettings(
177
326
  settings: ElevenLabsVoiceSettings,
178
327
  speedOverride: number | undefined,
@@ -194,6 +343,44 @@ function mapVoiceSettings(
194
343
  }
195
344
  }
196
345
 
346
+ /**
347
+ * ElevenLabs voice categories map onto {@link VoiceOrigin} with two joins:
348
+ * `famous` and `high_quality` are both curated tiers, so both read as
349
+ * `'professional'`. An unrecognized category is dropped rather than guessed,
350
+ * which keeps an `origins` filter from silently matching the wrong thing.
351
+ */
352
+ function toVoiceOrigin(
353
+ category: ElevenLabs.VoiceCategory | undefined,
354
+ ): VoiceOrigin | undefined {
355
+ switch (category) {
356
+ case 'premade':
357
+ return 'premade'
358
+ case 'generated':
359
+ return 'generated'
360
+ case 'cloned':
361
+ return 'cloned'
362
+ case 'professional':
363
+ case 'famous':
364
+ case 'high_quality':
365
+ return 'professional'
366
+ case undefined:
367
+ default:
368
+ return undefined
369
+ }
370
+ }
371
+
372
+ function toCatalogVoice(voice: ElevenLabs.Voice): CatalogVoice {
373
+ const origin = toVoiceOrigin(voice.category)
374
+ return {
375
+ voiceId: voice.voiceId,
376
+ ...(voice.name ? { name: voice.name } : {}),
377
+ ...(origin ? { origin } : {}),
378
+ ...(voice.description ? { description: voice.description } : {}),
379
+ ...(voice.previewUrl ? { previewUrl: voice.previewUrl } : {}),
380
+ ...(voice.labels ? { labels: voice.labels } : {}),
381
+ }
382
+ }
383
+
197
384
  /**
198
385
  * Map the standard TTSOptions `format` (mp3/opus/aac/flac/wav/pcm) to a
199
386
  * reasonable ElevenLabs `outputFormat` so callers don't need to know the
@@ -206,6 +393,7 @@ function inferOutputFormatFromResponseFormat(
206
393
  case 'mp3':
207
394
  return 'mp3_44100_128'
208
395
  case 'pcm':
396
+ case 'wav':
209
397
  return 'pcm_44100'
210
398
  case 'opus':
211
399
  return 'opus_48000_128'
@@ -213,14 +401,42 @@ function inferOutputFormatFromResponseFormat(
213
401
  return undefined
214
402
  case 'aac':
215
403
  case 'flac':
216
- case 'wav':
217
404
  default:
218
- // `aac` / `flac` / `wav` are not native ElevenLabs formats —
219
- // fall back to mp3 rather than blowing up mid-request.
220
- return 'mp3_44100_128'
405
+ throw new Error(
406
+ `ElevenLabs TTS does not support format '${format}'. Use mp3, pcm, opus, or wav.`,
407
+ )
221
408
  }
222
409
  }
223
410
 
411
+ function base64ToArrayBuffer(base64: string): ArrayBuffer {
412
+ const binary = atob(base64)
413
+ const bytes = new Uint8Array(binary.length)
414
+ for (let i = 0; i < binary.length; i += 1) bytes[i] = binary.charCodeAt(i)
415
+ return bytes.buffer
416
+ }
417
+
418
+ /** Wrap ElevenLabs pcm_44100 (16-bit little-endian mono) in a RIFF/WAV container. */
419
+ function wrapPcmAsWav(pcm: ArrayBuffer): ArrayBuffer {
420
+ const wav = new ArrayBuffer(44 + pcm.byteLength)
421
+ const bytes = new Uint8Array(wav)
422
+ const header = new DataView(wav)
423
+ const encoder = new TextEncoder()
424
+ bytes.set(encoder.encode('RIFF'), 0)
425
+ header.setUint32(4, 36 + pcm.byteLength, true)
426
+ bytes.set(encoder.encode('WAVEfmt '), 8)
427
+ header.setUint32(16, 16, true) // PCM format chunk size
428
+ header.setUint16(20, 1, true) // PCM encoding
429
+ header.setUint16(22, 1, true) // mono
430
+ header.setUint32(24, 44100, true) // sample rate
431
+ header.setUint32(28, 44100 * 2, true) // bytes per second
432
+ header.setUint16(32, 2, true) // bytes per sample
433
+ header.setUint16(34, 16, true) // bits per sample
434
+ bytes.set(encoder.encode('data'), 36)
435
+ header.setUint32(40, pcm.byteLength, true)
436
+ bytes.set(new Uint8Array(pcm), 44)
437
+ return wav
438
+ }
439
+
224
440
  /**
225
441
  * Create an ElevenLabs speech adapter using `ELEVENLABS_API_KEY` from env.
226
442
  */
@@ -0,0 +1,299 @@
1
+ import { BaseVoiceAdapter } from '@tanstack/ai/adapters'
2
+ import {
3
+ arrayBufferToBase64,
4
+ createElevenLabsClient,
5
+ generateId,
6
+ parseOutputFormat,
7
+ } from '../utils/client'
8
+ import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
9
+ import type {
10
+ GeneratedVoice,
11
+ VoiceGenerationOptions,
12
+ VoiceResult,
13
+ } from '@tanstack/ai'
14
+ import type { ElevenLabsClientConfig } from '../utils/client'
15
+ import type {
16
+ ElevenLabsOutputFormat,
17
+ ElevenLabsVoiceModel,
18
+ } from '../model-meta'
19
+
20
+ /**
21
+ * Provider-specific voice-design options. Fields map 1:1 onto the SDK's
22
+ * `VoiceDesignRequestModel` — mirroring the names so ElevenLabs'
23
+ * documentation stays useful here.
24
+ * @see https://elevenlabs.io/docs/api-reference/text-to-voice/design
25
+ */
26
+ export interface ElevenLabsVoiceProviderOptions {
27
+ /** Output audio format for the previews, `codec_samplerate[_bitrate]`. */
28
+ outputFormat?: ElevenLabsOutputFormat
29
+ /** Line the previews speak, 100..1000 characters. */
30
+ text?: string
31
+ /** Let ElevenLabs write the preview line from the description. */
32
+ autoGenerateText?: boolean
33
+ /** Preview loudness, -1 (quietest) to 1 (loudest). 0 is roughly -24 LUFS. */
34
+ loudness?: number
35
+ /** Deterministic sampling seed — same seed and inputs produce the same voice. */
36
+ seed?: number
37
+ /** How closely to follow the description. High values can sound robotic. */
38
+ guidanceScale?: number
39
+ /** Higher quality trades variety for fidelity. */
40
+ quality?: number
41
+ /** Let ElevenLabs expand a short description into a detailed one. */
42
+ shouldEnhance?: boolean
43
+ /**
44
+ * Balance of description against reference audio, 0 (almost all reference)
45
+ * to 1 (almost all description). `eleven_ttv_v3` only.
46
+ */
47
+ promptStrength?: number
48
+ /** Metadata stored on the voice. Only used when the voice is saved. */
49
+ labels?: Record<string, string>
50
+ /** Remixing session to attach these generations to. */
51
+ remixingSessionId?: string
52
+ /** Remixing session iteration to attach these generations to. */
53
+ remixingSessionIterationId?: string
54
+ }
55
+
56
+ /**
57
+ * ElevenLabs voice-design adapter built on the official
58
+ * `@elevenlabs/elevenlabs-js` SDK.
59
+ *
60
+ * ElevenLabs designs voices in two steps — generate previews, then promote one
61
+ * into a real voice — but that is an implementation detail. Pass `name` to get
62
+ * a saved voice back; leave it off to audition previews first.
63
+ *
64
+ * @example Audition previews
65
+ * ```ts
66
+ * const result = await generateVoice({
67
+ * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
68
+ * prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
69
+ * })
70
+ * ```
71
+ *
72
+ * @example Save the voice
73
+ * ```ts
74
+ * const result = await generateVoice({
75
+ * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
76
+ * prompt: 'A bright, upbeat product demo host',
77
+ * name: 'Demo Host',
78
+ * })
79
+ * ```
80
+ */
81
+ export class ElevenLabsVoiceAdapter<
82
+ TModel extends ElevenLabsVoiceModel,
83
+ > extends BaseVoiceAdapter<TModel, ElevenLabsVoiceProviderOptions> {
84
+ readonly name = 'elevenlabs' as const
85
+
86
+ private readonly client: ElevenLabsClient
87
+
88
+ constructor(model: TModel, config?: ElevenLabsClientConfig) {
89
+ super(model, config ?? {})
90
+ this.client = createElevenLabsClient(config)
91
+ }
92
+
93
+ async generateVoice(
94
+ options: VoiceGenerationOptions<ElevenLabsVoiceProviderOptions>,
95
+ ): Promise<VoiceResult> {
96
+ const { logger } = options
97
+ logger.request(
98
+ `activity=generateVoice provider=elevenlabs model=${this.model}`,
99
+ { provider: 'elevenlabs', model: this.model },
100
+ )
101
+ try {
102
+ // `voiceDescription` is required by the design endpoint, so reference
103
+ // audio alone is not enough here — unlike clone-only providers.
104
+ if (!options.prompt) {
105
+ throw new Error(
106
+ 'ElevenLabs voice design requires a `prompt` describing the voice. Reference audio alone is not supported; pass both to guide the design with a real speaker.',
107
+ )
108
+ }
109
+
110
+ const opts = options.modelOptions ?? {}
111
+ const referenceAudioBase64 = await toReferenceAudioBase64(
112
+ options.referenceAudio,
113
+ )
114
+ if (referenceAudioBase64 && this.model !== 'eleven_ttv_v3') {
115
+ throw new Error(
116
+ `ElevenLabs only accepts reference audio on eleven_ttv_v3, but this adapter is using "${this.model}".`,
117
+ )
118
+ }
119
+
120
+ const requestOptions = options.abortSignal
121
+ ? { abortSignal: options.abortSignal }
122
+ : {}
123
+
124
+ const previewResponse = await this.client.textToVoice.design(
125
+ {
126
+ voiceDescription: options.prompt,
127
+ modelId: this.model,
128
+ ...(opts.outputFormat ? { outputFormat: opts.outputFormat } : {}),
129
+ ...(opts.text ? { text: opts.text } : {}),
130
+ ...(opts.autoGenerateText != null
131
+ ? { autoGenerateText: opts.autoGenerateText }
132
+ : {}),
133
+ ...(opts.loudness != null ? { loudness: opts.loudness } : {}),
134
+ ...(opts.seed != null ? { seed: opts.seed } : {}),
135
+ ...(opts.guidanceScale != null
136
+ ? { guidanceScale: opts.guidanceScale }
137
+ : {}),
138
+ ...(opts.quality != null ? { quality: opts.quality } : {}),
139
+ ...(opts.shouldEnhance != null
140
+ ? { shouldEnhance: opts.shouldEnhance }
141
+ : {}),
142
+ ...(opts.promptStrength != null
143
+ ? { promptStrength: opts.promptStrength }
144
+ : {}),
145
+ ...(referenceAudioBase64 ? { referenceAudioBase64 } : {}),
146
+ ...(opts.remixingSessionId
147
+ ? { remixingSessionId: opts.remixingSessionId }
148
+ : {}),
149
+ ...(opts.remixingSessionIterationId
150
+ ? { remixingSessionIterationId: opts.remixingSessionIterationId }
151
+ : {}),
152
+ },
153
+ requestOptions,
154
+ )
155
+
156
+ const { format, contentType } = parseOutputFormat(opts.outputFormat)
157
+ const voices: Array<GeneratedVoice> = previewResponse.previews.map(
158
+ (preview) => ({
159
+ voiceId: preview.generatedVoiceId,
160
+ audio: preview.audioBase64,
161
+ format,
162
+ contentType: preview.mediaType || contentType,
163
+ duration: preview.durationSecs,
164
+ ...(preview.language ? { language: preview.language } : {}),
165
+ saved: false,
166
+ // The design endpoint returns a finished preview; nothing trains.
167
+ status: 'ready' as const,
168
+ }),
169
+ )
170
+
171
+ // A caller who passed `name` asked for a persisted voice. Returning an
172
+ // empty list would report success for a request that produced nothing.
173
+ if (voices.length === 0) {
174
+ throw new Error(
175
+ 'ElevenLabs returned no voice previews for this description. Try a longer, more specific prompt.',
176
+ )
177
+ }
178
+
179
+ if (options.name) {
180
+ voices[0] = await this.promoteFirstPreview(
181
+ voices,
182
+ options.name,
183
+ options.description ?? options.prompt,
184
+ opts.labels,
185
+ requestOptions,
186
+ )
187
+ }
188
+
189
+ return {
190
+ id: generateId(this.name),
191
+ model: this.model,
192
+ voices,
193
+ previewText: previewResponse.text,
194
+ }
195
+ } catch (error) {
196
+ logger.errors('elevenlabs.generateVoice fatal', {
197
+ error,
198
+ source: 'elevenlabs.generateVoice',
199
+ })
200
+ throw error
201
+ }
202
+ }
203
+
204
+ /**
205
+ * Promote the best preview into a real library voice. The remaining
206
+ * generated ids go along as `playedNotSelectedVoiceIds` — ElevenLabs uses
207
+ * them as RLHF signal for future designs.
208
+ */
209
+ private async promoteFirstPreview(
210
+ voices: ReadonlyArray<GeneratedVoice>,
211
+ voiceName: string,
212
+ voiceDescription: string,
213
+ labels: Record<string, string> | undefined,
214
+ requestOptions: { abortSignal?: AbortSignal },
215
+ ): Promise<GeneratedVoice> {
216
+ const [best, ...rest] = voices
217
+ if (!best) {
218
+ throw new Error('No preview to promote into a library voice.')
219
+ }
220
+
221
+ const saved = await this.client.textToVoice.create(
222
+ {
223
+ voiceName,
224
+ voiceDescription,
225
+ generatedVoiceId: best.voiceId,
226
+ ...(labels ? { labels } : {}),
227
+ ...(rest.length > 0
228
+ ? { playedNotSelectedVoiceIds: rest.map((voice) => voice.voiceId) }
229
+ : {}),
230
+ },
231
+ requestOptions,
232
+ )
233
+
234
+ return { ...best, voiceId: saved.voiceId, saved: true }
235
+ }
236
+
237
+ protected override generateId(): string {
238
+ return generateId(this.name)
239
+ }
240
+ }
241
+
242
+ /**
243
+ * Normalize reference audio to the bare base64 the design endpoint wants.
244
+ *
245
+ * https URLs are rejected rather than fetched: ElevenLabs has no URL field
246
+ * here, and downloading caller-supplied media into memory to inline it is a
247
+ * footgun on large files.
248
+ */
249
+ async function toReferenceAudioBase64(
250
+ audio: VoiceGenerationOptions['referenceAudio'],
251
+ ): Promise<string | undefined> {
252
+ if (audio == null) return undefined
253
+ if (audio instanceof ArrayBuffer) return arrayBufferToBase64(audio)
254
+ if (typeof audio !== 'string') {
255
+ return arrayBufferToBase64(await audio.arrayBuffer())
256
+ }
257
+
258
+ if (audio.startsWith('data:')) {
259
+ const commaIndex = audio.indexOf(',')
260
+ const header = commaIndex === -1 ? '' : audio.slice(5, commaIndex)
261
+ if (commaIndex === -1 || !/;base64$/i.test(header)) {
262
+ throw new Error(
263
+ 'ElevenLabs voice design needs base64 reference audio. Pass a base64 data URL, a base64 string, a Blob, or an ArrayBuffer.',
264
+ )
265
+ }
266
+ return audio.slice(commaIndex + 1)
267
+ }
268
+
269
+ if (/^https?:\/\//i.test(audio)) {
270
+ throw new Error(
271
+ 'ElevenLabs voice design does not accept reference audio URLs. Read the file yourself and pass a Blob, ArrayBuffer, or base64 string.',
272
+ )
273
+ }
274
+
275
+ return audio
276
+ }
277
+
278
+ /**
279
+ * Create an ElevenLabs voice-design adapter using `ELEVENLABS_API_KEY` from env.
280
+ */
281
+ export function elevenlabsVoiceDesign<TModel extends ElevenLabsVoiceModel>(
282
+ model: TModel,
283
+ config?: ElevenLabsClientConfig,
284
+ ): ElevenLabsVoiceAdapter<TModel> {
285
+ return new ElevenLabsVoiceAdapter(model, config)
286
+ }
287
+
288
+ /**
289
+ * Create an ElevenLabs voice-design adapter with an explicit API key.
290
+ */
291
+ export function createElevenLabsVoiceDesign<
292
+ TModel extends ElevenLabsVoiceModel,
293
+ >(
294
+ model: TModel,
295
+ apiKey: string,
296
+ config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
297
+ ): ElevenLabsVoiceAdapter<TModel> {
298
+ return new ElevenLabsVoiceAdapter(model, { apiKey, ...config })
299
+ }
package/src/index.ts CHANGED
@@ -38,6 +38,17 @@ export {
38
38
  type ElevenLabsMusicCompositionPlan,
39
39
  } from './adapters/audio'
40
40
 
41
+ // ============================================================================
42
+ // Voice (Voice Design) Adapter
43
+ // ============================================================================
44
+
45
+ export {
46
+ ElevenLabsVoiceAdapter,
47
+ createElevenLabsVoiceDesign,
48
+ elevenlabsVoiceDesign,
49
+ type ElevenLabsVoiceProviderOptions,
50
+ } from './adapters/voice'
51
+
41
52
  // ============================================================================
42
53
  // Transcription (Speech-to-Text) Adapter
43
54
  // ============================================================================
@@ -57,6 +68,7 @@ export {
57
68
  ELEVENLABS_TTS_MODELS,
58
69
  ELEVENLABS_AUDIO_MODELS,
59
70
  ELEVENLABS_TRANSCRIPTION_MODELS,
71
+ ELEVENLABS_VOICE_MODELS,
60
72
  isElevenLabsMusicModel,
61
73
  isElevenLabsSoundEffectsModel,
62
74
  type ElevenLabsTTSModel,
@@ -64,6 +76,7 @@ export {
64
76
  type ElevenLabsMusicModel,
65
77
  type ElevenLabsSoundEffectsModel,
66
78
  type ElevenLabsTranscriptionModel,
79
+ type ElevenLabsVoiceModel,
67
80
  type ElevenLabsOutputFormat,
68
81
  } from './model-meta'
69
82