@tanstack/ai-elevenlabs 0.4.4 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,8 +6,18 @@ import {
6
6
  parseOutputFormat,
7
7
  readStreamToArrayBuffer,
8
8
  } from '../utils/client'
9
- import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
10
- import type { TTSOptions, TTSResult } from '@tanstack/ai'
9
+ import type { ElevenLabs, ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
10
+ import type {
11
+ CatalogVoice,
12
+ ListVoicesOptions,
13
+ ListVoicesResult,
14
+ TTSAlignment,
15
+ TTSCapabilities,
16
+ TTSOptions,
17
+ TTSResult,
18
+ TTSSegment,
19
+ VoiceOrigin,
20
+ } from '@tanstack/ai'
11
21
  import type { ElevenLabsClientConfig } from '../utils/client'
12
22
  import type { ElevenLabsOutputFormat, ElevenLabsTTSModel } from '../model-meta'
13
23
 
@@ -82,6 +92,15 @@ export class ElevenLabsSpeechAdapter<
82
92
  > extends BaseTTSAdapter<TModel, ElevenLabsSpeechProviderOptions> {
83
93
  readonly name = 'elevenlabs' as const
84
94
 
95
+ /**
96
+ * `textToDialogue` accepts up to 10 distinct voice ids, and both endpoints
97
+ * have a `…WithTimestamps` twin that returns character alignment.
98
+ */
99
+ override readonly capabilities: TTSCapabilities = {
100
+ maxSpeakers: 10,
101
+ timestamps: true,
102
+ }
103
+
85
104
  private readonly client: ElevenLabsClient
86
105
 
87
106
  constructor(model: TModel, config?: ElevenLabsClientConfig) {
@@ -98,12 +117,6 @@ export class ElevenLabsSpeechAdapter<
98
117
  { provider: 'elevenlabs', model: this.model },
99
118
  )
100
119
  try {
101
- const voiceId = options.voice ?? options.modelOptions?.voiceId
102
- if (!voiceId) {
103
- throw new Error(
104
- 'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
105
- )
106
- }
107
120
  const {
108
121
  outputFormat,
109
122
  voiceSettings,
@@ -121,46 +134,115 @@ export class ElevenLabsSpeechAdapter<
121
134
  const effectiveOutputFormat =
122
135
  outputFormat ?? inferOutputFormatFromResponseFormat(options.format)
123
136
  const wrapAsWav = outputFormat == null && options.format === 'wav'
124
-
125
- const stream = await this.client.textToSpeech.convert(voiceId, {
126
- text: options.text,
137
+ const { format, contentType } = wrapAsWav
138
+ ? { format: 'wav', contentType: 'audio/wav' }
139
+ : parseOutputFormat(effectiveOutputFormat)
140
+ // `format: 'wav'` asks ElevenLabs for pcm_44100 and wraps it here, on
141
+ // every endpoint: the byte-stream pair hands back raw bytes, the
142
+ // timestamped pair hands back the same bytes as base64.
143
+ const encodeAudio = (buffer: ArrayBuffer) =>
144
+ arrayBufferToBase64(wrapAsWav ? wrapPcmAsWav(buffer) : buffer)
145
+ const encodeAudioBase64 = (audioBase64: string) =>
146
+ wrapAsWav ? encodeAudio(base64ToArrayBuffer(audioBase64)) : audioBase64
147
+ const base = {
127
148
  modelId: this.model,
128
149
  ...(effectiveOutputFormat
129
150
  ? { outputFormat: effectiveOutputFormat }
130
151
  : {}),
152
+ ...(languageCode ? { languageCode } : {}),
153
+ ...(seed != null ? { seed } : {}),
154
+ ...(applyTextNormalization ? { applyTextNormalization } : {}),
155
+ }
156
+
157
+ // Four endpoints, picked by (turns?, timestamps?). The dialogue pair
158
+ // takes `inputs` instead of a voice id in the path, and the timestamped
159
+ // pair returns JSON (base64 audio + alignment) instead of a byte stream.
160
+ if (options.turns) {
161
+ const inputs = options.turns.map((turn) => ({
162
+ text: turn.text,
163
+ voiceId: turn.voice,
164
+ }))
165
+ const dialogue = {
166
+ ...base,
167
+ inputs,
168
+ ...(voiceSettings?.stability != null
169
+ ? { settings: { stability: voiceSettings.stability } }
170
+ : {}),
171
+ }
172
+
173
+ if (options.timestamps) {
174
+ const response =
175
+ await this.client.textToDialogue.convertWithTimestamps(dialogue)
176
+ return {
177
+ id: generateId(this.name),
178
+ model: this.model,
179
+ audio: encodeAudioBase64(response.audioBase64),
180
+ format,
181
+ contentType,
182
+ ...toAlignmentFields(response.alignment, response.voiceSegments),
183
+ }
184
+ }
185
+
186
+ const stream = await this.client.textToDialogue.convert(dialogue)
187
+ return {
188
+ id: generateId(this.name),
189
+ model: this.model,
190
+ audio: encodeAudio(await readStreamToArrayBuffer(stream)),
191
+ format,
192
+ contentType,
193
+ }
194
+ }
195
+
196
+ const voiceId = options.voice ?? options.modelOptions?.voiceId
197
+ if (!voiceId) {
198
+ throw new Error(
199
+ 'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
200
+ )
201
+ }
202
+ const single = {
203
+ ...base,
204
+ text: options.text,
131
205
  ...(voiceSettings
132
206
  ? { voiceSettings: mapVoiceSettings(voiceSettings, options.speed) }
133
207
  : options.speed != null
134
208
  ? { voiceSettings: { speed: options.speed } }
135
209
  : {}),
136
- ...(languageCode ? { languageCode } : {}),
137
- ...(seed != null ? { seed } : {}),
138
210
  ...(previousText ? { previousText } : {}),
139
211
  ...(nextText ? { nextText } : {}),
140
212
  ...(previousRequestIds ? { previousRequestIds } : {}),
141
213
  ...(nextRequestIds ? { nextRequestIds } : {}),
142
- ...(applyTextNormalization ? { applyTextNormalization } : {}),
143
214
  ...(applyLanguageTextNormalization != null
144
215
  ? { applyLanguageTextNormalization }
145
216
  : {}),
217
+ ...(enableLogging != null ? { enableLogging } : {}),
218
+ }
219
+
220
+ if (options.timestamps) {
221
+ const response = await this.client.textToSpeech.convertWithTimestamps(
222
+ voiceId,
223
+ single,
224
+ )
225
+ return {
226
+ id: generateId(this.name),
227
+ model: this.model,
228
+ audio: encodeAudioBase64(response.audioBase64),
229
+ format,
230
+ contentType,
231
+ ...toAlignmentFields(response.alignment, undefined),
232
+ }
233
+ }
234
+
235
+ const stream = await this.client.textToSpeech.convert(voiceId, {
236
+ ...single,
146
237
  ...(optimizeStreamingLatency != null
147
238
  ? { optimizeStreamingLatency }
148
239
  : {}),
149
- ...(enableLogging != null ? { enableLogging } : {}),
150
240
  })
151
241
 
152
- const buffer = await readStreamToArrayBuffer(stream)
153
- const base64 = arrayBufferToBase64(
154
- wrapAsWav ? wrapPcmAsWav(buffer) : buffer,
155
- )
156
- const { format, contentType } = wrapAsWav
157
- ? { format: 'wav', contentType: 'audio/wav' }
158
- : parseOutputFormat(effectiveOutputFormat)
159
-
160
242
  return {
161
243
  id: generateId(this.name),
162
244
  model: this.model,
163
- audio: base64,
245
+ audio: encodeAudio(await readStreamToArrayBuffer(stream)),
164
246
  format,
165
247
  contentType,
166
248
  }
@@ -173,11 +255,73 @@ export class ElevenLabsSpeechAdapter<
173
255
  }
174
256
  }
175
257
 
258
+ /**
259
+ * List the voices this API key can use, via `GET /v1/voices`.
260
+ *
261
+ * The catalog is per-account and grows every time `generateVoice()` saves a
262
+ * voice, so it has to be read at runtime rather than shipped as a const.
263
+ */
264
+ override async listVoices(
265
+ options?: ListVoicesOptions,
266
+ ): Promise<ListVoicesResult> {
267
+ const response = await this.client.voices.getAll(
268
+ {},
269
+ options?.abortSignal ? { abortSignal: options.abortSignal } : {},
270
+ )
271
+
272
+ const voices = response.voices.map(toCatalogVoice)
273
+ const origins = options?.origins
274
+ // ElevenLabs has no origin filter on /v1/voices, so narrow in memory.
275
+ return {
276
+ voices: origins
277
+ ? voices.filter(
278
+ (voice) => voice.origin && origins.includes(voice.origin),
279
+ )
280
+ : voices,
281
+ }
282
+ }
283
+
176
284
  protected override generateId(): string {
177
285
  return generateId(this.name)
178
286
  }
179
287
  }
180
288
 
289
+ /**
290
+ * Map the ElevenLabs timestamp payload onto the core `alignment` / `segments`
291
+ * fields. `voiceSegments` only comes back from the dialogue endpoint; its
292
+ * character indices slice the alignment into per-turn text.
293
+ */
294
+ function toAlignmentFields(
295
+ alignment: ElevenLabs.CharacterAlignmentResponseModel | undefined,
296
+ voiceSegments: Array<ElevenLabs.VoiceSegment> | undefined,
297
+ ): { alignment?: TTSAlignment; segments?: Array<TTSSegment> } {
298
+ const fields: { alignment?: TTSAlignment; segments?: Array<TTSSegment> } = {}
299
+ if (alignment) {
300
+ fields.alignment = {
301
+ unit: 'character',
302
+ texts: alignment.characters,
303
+ startSeconds: alignment.characterStartTimesSeconds,
304
+ endSeconds: alignment.characterEndTimesSeconds,
305
+ }
306
+ }
307
+ if (voiceSegments && voiceSegments.length > 0) {
308
+ fields.segments = voiceSegments.map((segment) => ({
309
+ startSeconds: segment.startTimeSeconds,
310
+ endSeconds: segment.endTimeSeconds,
311
+ turnIndex: segment.dialogueInputIndex,
312
+ voice: segment.voiceId,
313
+ ...(alignment
314
+ ? {
315
+ text: alignment.characters
316
+ .slice(segment.characterStartIndex, segment.characterEndIndex)
317
+ .join(''),
318
+ }
319
+ : {}),
320
+ }))
321
+ }
322
+ return fields
323
+ }
324
+
181
325
  function mapVoiceSettings(
182
326
  settings: ElevenLabsVoiceSettings,
183
327
  speedOverride: number | undefined,
@@ -199,6 +343,44 @@ function mapVoiceSettings(
199
343
  }
200
344
  }
201
345
 
346
+ /**
347
+ * ElevenLabs voice categories map onto {@link VoiceOrigin} with two joins:
348
+ * `famous` and `high_quality` are both curated tiers, so both read as
349
+ * `'professional'`. An unrecognized category is dropped rather than guessed,
350
+ * which keeps an `origins` filter from silently matching the wrong thing.
351
+ */
352
+ function toVoiceOrigin(
353
+ category: ElevenLabs.VoiceCategory | undefined,
354
+ ): VoiceOrigin | undefined {
355
+ switch (category) {
356
+ case 'premade':
357
+ return 'premade'
358
+ case 'generated':
359
+ return 'generated'
360
+ case 'cloned':
361
+ return 'cloned'
362
+ case 'professional':
363
+ case 'famous':
364
+ case 'high_quality':
365
+ return 'professional'
366
+ case undefined:
367
+ default:
368
+ return undefined
369
+ }
370
+ }
371
+
372
+ function toCatalogVoice(voice: ElevenLabs.Voice): CatalogVoice {
373
+ const origin = toVoiceOrigin(voice.category)
374
+ return {
375
+ voiceId: voice.voiceId,
376
+ ...(voice.name ? { name: voice.name } : {}),
377
+ ...(origin ? { origin } : {}),
378
+ ...(voice.description ? { description: voice.description } : {}),
379
+ ...(voice.previewUrl ? { previewUrl: voice.previewUrl } : {}),
380
+ ...(voice.labels ? { labels: voice.labels } : {}),
381
+ }
382
+ }
383
+
202
384
  /**
203
385
  * Map the standard TTSOptions `format` (mp3/opus/aac/flac/wav/pcm) to a
204
386
  * reasonable ElevenLabs `outputFormat` so callers don't need to know the
@@ -226,6 +408,13 @@ function inferOutputFormatFromResponseFormat(
226
408
  }
227
409
  }
228
410
 
411
+ function base64ToArrayBuffer(base64: string): ArrayBuffer {
412
+ const binary = atob(base64)
413
+ const bytes = new Uint8Array(binary.length)
414
+ for (let i = 0; i < binary.length; i += 1) bytes[i] = binary.charCodeAt(i)
415
+ return bytes.buffer
416
+ }
417
+
229
418
  /** Wrap ElevenLabs pcm_44100 (16-bit little-endian mono) in a RIFF/WAV container. */
230
419
  function wrapPcmAsWav(pcm: ArrayBuffer): ArrayBuffer {
231
420
  const wav = new ArrayBuffer(44 + pcm.byteLength)
@@ -0,0 +1,299 @@
1
+ import { BaseVoiceAdapter } from '@tanstack/ai/adapters'
2
+ import {
3
+ arrayBufferToBase64,
4
+ createElevenLabsClient,
5
+ generateId,
6
+ parseOutputFormat,
7
+ } from '../utils/client'
8
+ import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
9
+ import type {
10
+ GeneratedVoice,
11
+ VoiceGenerationOptions,
12
+ VoiceResult,
13
+ } from '@tanstack/ai'
14
+ import type { ElevenLabsClientConfig } from '../utils/client'
15
+ import type {
16
+ ElevenLabsOutputFormat,
17
+ ElevenLabsVoiceModel,
18
+ } from '../model-meta'
19
+
20
+ /**
21
+ * Provider-specific voice-design options. Fields map 1:1 onto the SDK's
22
+ * `VoiceDesignRequestModel` — mirroring the names so ElevenLabs'
23
+ * documentation stays useful here.
24
+ * @see https://elevenlabs.io/docs/api-reference/text-to-voice/design
25
+ */
26
+ export interface ElevenLabsVoiceProviderOptions {
27
+ /** Output audio format for the previews, `codec_samplerate[_bitrate]`. */
28
+ outputFormat?: ElevenLabsOutputFormat
29
+ /** Line the previews speak, 100..1000 characters. */
30
+ text?: string
31
+ /** Let ElevenLabs write the preview line from the description. */
32
+ autoGenerateText?: boolean
33
+ /** Preview loudness, -1 (quietest) to 1 (loudest). 0 is roughly -24 LUFS. */
34
+ loudness?: number
35
+ /** Deterministic sampling seed — same seed and inputs produce the same voice. */
36
+ seed?: number
37
+ /** How closely to follow the description. High values can sound robotic. */
38
+ guidanceScale?: number
39
+ /** Higher quality trades variety for fidelity. */
40
+ quality?: number
41
+ /** Let ElevenLabs expand a short description into a detailed one. */
42
+ shouldEnhance?: boolean
43
+ /**
44
+ * Balance of description against reference audio, 0 (almost all reference)
45
+ * to 1 (almost all description). `eleven_ttv_v3` only.
46
+ */
47
+ promptStrength?: number
48
+ /** Metadata stored on the voice. Only used when the voice is saved. */
49
+ labels?: Record<string, string>
50
+ /** Remixing session to attach these generations to. */
51
+ remixingSessionId?: string
52
+ /** Remixing session iteration to attach these generations to. */
53
+ remixingSessionIterationId?: string
54
+ }
55
+
56
+ /**
57
+ * ElevenLabs voice-design adapter built on the official
58
+ * `@elevenlabs/elevenlabs-js` SDK.
59
+ *
60
+ * ElevenLabs designs voices in two steps — generate previews, then promote one
61
+ * into a real voice — but that is an implementation detail. Pass `name` to get
62
+ * a saved voice back; leave it off to audition previews first.
63
+ *
64
+ * @example Audition previews
65
+ * ```ts
66
+ * const result = await generateVoice({
67
+ * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
68
+ * prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
69
+ * })
70
+ * ```
71
+ *
72
+ * @example Save the voice
73
+ * ```ts
74
+ * const result = await generateVoice({
75
+ * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
76
+ * prompt: 'A bright, upbeat product demo host',
77
+ * name: 'Demo Host',
78
+ * })
79
+ * ```
80
+ */
81
+ export class ElevenLabsVoiceAdapter<
82
+ TModel extends ElevenLabsVoiceModel,
83
+ > extends BaseVoiceAdapter<TModel, ElevenLabsVoiceProviderOptions> {
84
+ readonly name = 'elevenlabs' as const
85
+
86
+ private readonly client: ElevenLabsClient
87
+
88
+ constructor(model: TModel, config?: ElevenLabsClientConfig) {
89
+ super(model, config ?? {})
90
+ this.client = createElevenLabsClient(config)
91
+ }
92
+
93
+ async generateVoice(
94
+ options: VoiceGenerationOptions<ElevenLabsVoiceProviderOptions>,
95
+ ): Promise<VoiceResult> {
96
+ const { logger } = options
97
+ logger.request(
98
+ `activity=generateVoice provider=elevenlabs model=${this.model}`,
99
+ { provider: 'elevenlabs', model: this.model },
100
+ )
101
+ try {
102
+ // `voiceDescription` is required by the design endpoint, so reference
103
+ // audio alone is not enough here — unlike clone-only providers.
104
+ if (!options.prompt) {
105
+ throw new Error(
106
+ 'ElevenLabs voice design requires a `prompt` describing the voice. Reference audio alone is not supported; pass both to guide the design with a real speaker.',
107
+ )
108
+ }
109
+
110
+ const opts = options.modelOptions ?? {}
111
+ const referenceAudioBase64 = await toReferenceAudioBase64(
112
+ options.referenceAudio,
113
+ )
114
+ if (referenceAudioBase64 && this.model !== 'eleven_ttv_v3') {
115
+ throw new Error(
116
+ `ElevenLabs only accepts reference audio on eleven_ttv_v3, but this adapter is using "${this.model}".`,
117
+ )
118
+ }
119
+
120
+ const requestOptions = options.abortSignal
121
+ ? { abortSignal: options.abortSignal }
122
+ : {}
123
+
124
+ const previewResponse = await this.client.textToVoice.design(
125
+ {
126
+ voiceDescription: options.prompt,
127
+ modelId: this.model,
128
+ ...(opts.outputFormat ? { outputFormat: opts.outputFormat } : {}),
129
+ ...(opts.text ? { text: opts.text } : {}),
130
+ ...(opts.autoGenerateText != null
131
+ ? { autoGenerateText: opts.autoGenerateText }
132
+ : {}),
133
+ ...(opts.loudness != null ? { loudness: opts.loudness } : {}),
134
+ ...(opts.seed != null ? { seed: opts.seed } : {}),
135
+ ...(opts.guidanceScale != null
136
+ ? { guidanceScale: opts.guidanceScale }
137
+ : {}),
138
+ ...(opts.quality != null ? { quality: opts.quality } : {}),
139
+ ...(opts.shouldEnhance != null
140
+ ? { shouldEnhance: opts.shouldEnhance }
141
+ : {}),
142
+ ...(opts.promptStrength != null
143
+ ? { promptStrength: opts.promptStrength }
144
+ : {}),
145
+ ...(referenceAudioBase64 ? { referenceAudioBase64 } : {}),
146
+ ...(opts.remixingSessionId
147
+ ? { remixingSessionId: opts.remixingSessionId }
148
+ : {}),
149
+ ...(opts.remixingSessionIterationId
150
+ ? { remixingSessionIterationId: opts.remixingSessionIterationId }
151
+ : {}),
152
+ },
153
+ requestOptions,
154
+ )
155
+
156
+ const { format, contentType } = parseOutputFormat(opts.outputFormat)
157
+ const voices: Array<GeneratedVoice> = previewResponse.previews.map(
158
+ (preview) => ({
159
+ voiceId: preview.generatedVoiceId,
160
+ audio: preview.audioBase64,
161
+ format,
162
+ contentType: preview.mediaType || contentType,
163
+ duration: preview.durationSecs,
164
+ ...(preview.language ? { language: preview.language } : {}),
165
+ saved: false,
166
+ // The design endpoint returns a finished preview; nothing trains.
167
+ status: 'ready' as const,
168
+ }),
169
+ )
170
+
171
+ // A caller who passed `name` asked for a persisted voice. Returning an
172
+ // empty list would report success for a request that produced nothing.
173
+ if (voices.length === 0) {
174
+ throw new Error(
175
+ 'ElevenLabs returned no voice previews for this description. Try a longer, more specific prompt.',
176
+ )
177
+ }
178
+
179
+ if (options.name) {
180
+ voices[0] = await this.promoteFirstPreview(
181
+ voices,
182
+ options.name,
183
+ options.description ?? options.prompt,
184
+ opts.labels,
185
+ requestOptions,
186
+ )
187
+ }
188
+
189
+ return {
190
+ id: generateId(this.name),
191
+ model: this.model,
192
+ voices,
193
+ previewText: previewResponse.text,
194
+ }
195
+ } catch (error) {
196
+ logger.errors('elevenlabs.generateVoice fatal', {
197
+ error,
198
+ source: 'elevenlabs.generateVoice',
199
+ })
200
+ throw error
201
+ }
202
+ }
203
+
204
+ /**
205
+ * Promote the best preview into a real library voice. The remaining
206
+ * generated ids go along as `playedNotSelectedVoiceIds` — ElevenLabs uses
207
+ * them as RLHF signal for future designs.
208
+ */
209
+ private async promoteFirstPreview(
210
+ voices: ReadonlyArray<GeneratedVoice>,
211
+ voiceName: string,
212
+ voiceDescription: string,
213
+ labels: Record<string, string> | undefined,
214
+ requestOptions: { abortSignal?: AbortSignal },
215
+ ): Promise<GeneratedVoice> {
216
+ const [best, ...rest] = voices
217
+ if (!best) {
218
+ throw new Error('No preview to promote into a library voice.')
219
+ }
220
+
221
+ const saved = await this.client.textToVoice.create(
222
+ {
223
+ voiceName,
224
+ voiceDescription,
225
+ generatedVoiceId: best.voiceId,
226
+ ...(labels ? { labels } : {}),
227
+ ...(rest.length > 0
228
+ ? { playedNotSelectedVoiceIds: rest.map((voice) => voice.voiceId) }
229
+ : {}),
230
+ },
231
+ requestOptions,
232
+ )
233
+
234
+ return { ...best, voiceId: saved.voiceId, saved: true }
235
+ }
236
+
237
+ protected override generateId(): string {
238
+ return generateId(this.name)
239
+ }
240
+ }
241
+
242
+ /**
243
+ * Normalize reference audio to the bare base64 the design endpoint wants.
244
+ *
245
+ * https URLs are rejected rather than fetched: ElevenLabs has no URL field
246
+ * here, and downloading caller-supplied media into memory to inline it is a
247
+ * footgun on large files.
248
+ */
249
+ async function toReferenceAudioBase64(
250
+ audio: VoiceGenerationOptions['referenceAudio'],
251
+ ): Promise<string | undefined> {
252
+ if (audio == null) return undefined
253
+ if (audio instanceof ArrayBuffer) return arrayBufferToBase64(audio)
254
+ if (typeof audio !== 'string') {
255
+ return arrayBufferToBase64(await audio.arrayBuffer())
256
+ }
257
+
258
+ if (audio.startsWith('data:')) {
259
+ const commaIndex = audio.indexOf(',')
260
+ const header = commaIndex === -1 ? '' : audio.slice(5, commaIndex)
261
+ if (commaIndex === -1 || !/;base64$/i.test(header)) {
262
+ throw new Error(
263
+ 'ElevenLabs voice design needs base64 reference audio. Pass a base64 data URL, a base64 string, a Blob, or an ArrayBuffer.',
264
+ )
265
+ }
266
+ return audio.slice(commaIndex + 1)
267
+ }
268
+
269
+ if (/^https?:\/\//i.test(audio)) {
270
+ throw new Error(
271
+ 'ElevenLabs voice design does not accept reference audio URLs. Read the file yourself and pass a Blob, ArrayBuffer, or base64 string.',
272
+ )
273
+ }
274
+
275
+ return audio
276
+ }
277
+
278
+ /**
279
+ * Create an ElevenLabs voice-design adapter using `ELEVENLABS_API_KEY` from env.
280
+ */
281
+ export function elevenlabsVoiceDesign<TModel extends ElevenLabsVoiceModel>(
282
+ model: TModel,
283
+ config?: ElevenLabsClientConfig,
284
+ ): ElevenLabsVoiceAdapter<TModel> {
285
+ return new ElevenLabsVoiceAdapter(model, config)
286
+ }
287
+
288
+ /**
289
+ * Create an ElevenLabs voice-design adapter with an explicit API key.
290
+ */
291
+ export function createElevenLabsVoiceDesign<
292
+ TModel extends ElevenLabsVoiceModel,
293
+ >(
294
+ model: TModel,
295
+ apiKey: string,
296
+ config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
297
+ ): ElevenLabsVoiceAdapter<TModel> {
298
+ return new ElevenLabsVoiceAdapter(model, { apiKey, ...config })
299
+ }
package/src/index.ts CHANGED
@@ -38,6 +38,17 @@ export {
38
38
  type ElevenLabsMusicCompositionPlan,
39
39
  } from './adapters/audio'
40
40
 
41
+ // ============================================================================
42
+ // Voice (Voice Design) Adapter
43
+ // ============================================================================
44
+
45
+ export {
46
+ ElevenLabsVoiceAdapter,
47
+ createElevenLabsVoiceDesign,
48
+ elevenlabsVoiceDesign,
49
+ type ElevenLabsVoiceProviderOptions,
50
+ } from './adapters/voice'
51
+
41
52
  // ============================================================================
42
53
  // Transcription (Speech-to-Text) Adapter
43
54
  // ============================================================================
@@ -57,6 +68,7 @@ export {
57
68
  ELEVENLABS_TTS_MODELS,
58
69
  ELEVENLABS_AUDIO_MODELS,
59
70
  ELEVENLABS_TRANSCRIPTION_MODELS,
71
+ ELEVENLABS_VOICE_MODELS,
60
72
  isElevenLabsMusicModel,
61
73
  isElevenLabsSoundEffectsModel,
62
74
  type ElevenLabsTTSModel,
@@ -64,6 +76,7 @@ export {
64
76
  type ElevenLabsMusicModel,
65
77
  type ElevenLabsSoundEffectsModel,
66
78
  type ElevenLabsTranscriptionModel,
79
+ type ElevenLabsVoiceModel,
67
80
  type ElevenLabsOutputFormat,
68
81
  } from './model-meta'
69
82