@tanstack/ai-elevenlabs 0.1.8 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/esm/adapters/audio.d.ts +93 -0
  2. package/dist/esm/adapters/audio.js +106 -0
  3. package/dist/esm/adapters/audio.js.map +1 -0
  4. package/dist/esm/adapters/speech.d.ts +83 -0
  5. package/dist/esm/adapters/speech.js +109 -0
  6. package/dist/esm/adapters/speech.js.map +1 -0
  7. package/dist/esm/adapters/transcription.d.ts +79 -0
  8. package/dist/esm/adapters/transcription.js +142 -0
  9. package/dist/esm/adapters/transcription.js.map +1 -0
  10. package/dist/esm/index.d.ts +5 -0
  11. package/dist/esm/index.js +21 -1
  12. package/dist/esm/index.js.map +1 -1
  13. package/dist/esm/model-meta.d.ts +42 -0
  14. package/dist/esm/model-meta.js +32 -0
  15. package/dist/esm/model-meta.js.map +1 -0
  16. package/dist/esm/realtime/adapter.d.ts +1 -1
  17. package/dist/esm/realtime/adapter.js +1 -1
  18. package/dist/esm/realtime/adapter.js.map +1 -1
  19. package/dist/esm/realtime/token.d.ts +11 -7
  20. package/dist/esm/realtime/token.js +8 -32
  21. package/dist/esm/realtime/token.js.map +1 -1
  22. package/dist/esm/realtime/types.d.ts +5 -2
  23. package/dist/esm/utils/client.d.ts +62 -0
  24. package/dist/esm/utils/client.js +124 -0
  25. package/dist/esm/utils/client.js.map +1 -0
  26. package/dist/esm/utils/index.d.ts +1 -0
  27. package/package.json +11 -3
  28. package/src/adapters/audio.ts +257 -0
  29. package/src/adapters/speech.ts +240 -0
  30. package/src/adapters/transcription.ts +338 -0
  31. package/src/index.ts +64 -0
  32. package/src/model-meta.ts +79 -0
  33. package/src/realtime/adapter.ts +3 -3
  34. package/src/realtime/token.ts +21 -56
  35. package/src/realtime/types.ts +5 -2
  36. package/src/utils/client.ts +196 -0
  37. package/src/utils/index.ts +10 -0
@@ -0,0 +1,257 @@
1
+ import { BaseAudioAdapter } from '@tanstack/ai/adapters'
2
+ import {
3
+ arrayBufferToBase64,
4
+ createElevenLabsClient,
5
+ generateId,
6
+ parseOutputFormat,
7
+ readStreamToArrayBuffer,
8
+ } from '../utils/client'
9
+ import {
10
+ isElevenLabsMusicModel,
11
+ isElevenLabsSoundEffectsModel,
12
+ } from '../model-meta'
13
+ import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
14
+ import type {
15
+ AudioGenerationOptions,
16
+ AudioGenerationResult,
17
+ } from '@tanstack/ai'
18
+ import type { ElevenLabsClientConfig } from '../utils/client'
19
+ import type {
20
+ ElevenLabsAudioModel,
21
+ ElevenLabsMusicModel,
22
+ ElevenLabsOutputFormat,
23
+ ElevenLabsSoundEffectsModel,
24
+ } from '../model-meta'
25
+
26
+ /**
27
+ * Structured composition plan for ElevenLabs music generation. Mutually
28
+ * exclusive with a free-form `prompt` on the `generateAudio()` call — when
29
+ * supplied, `prompt` is ignored by ElevenLabs.
30
+ *
31
+ * We mirror the SDK's camelCase naming. Lengths are in milliseconds.
32
+ * @see https://elevenlabs.io/docs/api-reference/music/compose
33
+ */
34
+ export interface ElevenLabsMusicCompositionPlan {
35
+ /** Positive global style descriptors (mood, instruments, tempo, …). */
36
+ positiveGlobalStyles?: Array<string>
37
+ /** Negative global style descriptors — styles to avoid. */
38
+ negativeGlobalStyles?: Array<string>
39
+ /** Section definitions (verse/chorus/bridge/…) with local style hints. */
40
+ sections?: Array<{
41
+ sectionName: string
42
+ positiveLocalStyles?: Array<string>
43
+ negativeLocalStyles?: Array<string>
44
+ durationMs?: number
45
+ lines?: Array<string>
46
+ }>
47
+ }
48
+
49
+ /**
50
+ * Provider options common to all ElevenLabs audio endpoints.
51
+ */
52
+ interface CommonAudioOptions {
53
+ /** Output audio format. Defaults to `mp3_44100_128`. */
54
+ outputFormat?: ElevenLabsOutputFormat
55
+ }
56
+
57
+ /**
58
+ * Provider options for music generation (`music_v1`).
59
+ */
60
+ export interface ElevenLabsMusicProviderOptions extends CommonAudioOptions {
61
+ /** Structured composition plan. Mutually exclusive with `prompt`/`duration`. */
62
+ compositionPlan?: ElevenLabsMusicCompositionPlan
63
+ /** Deterministic sampling seed (incompatible with `prompt`). */
64
+ seed?: number
65
+ /** Force the output to be purely instrumental (prompt-mode only). */
66
+ forceInstrumental?: boolean
67
+ /** Strictly respect section durations in `compositionPlan`. */
68
+ respectSectionsDurations?: boolean
69
+ }
70
+
71
+ /**
72
+ * Provider options for sound-effect generation (`eleven_text_to_sound_v*`).
73
+ */
74
+ export interface ElevenLabsSoundEffectsProviderOptions extends CommonAudioOptions {
75
+ /** Prompt influence, 0..1. Default 0.3. Higher = more prompt adherence. */
76
+ promptInfluence?: number
77
+ /** Generate a loopable SFX (v2 only). */
78
+ loop?: boolean
79
+ }
80
+
81
+ /**
82
+ * Union of per-model provider options. We keep both branches on one type so
83
+ * the adapter stays tree-shakeable; callers narrow by model at the factory.
84
+ */
85
+ export type ElevenLabsAudioProviderOptions =
86
+ | (ElevenLabsMusicProviderOptions & ElevenLabsSoundEffectsProviderOptions)
87
+ | ElevenLabsMusicProviderOptions
88
+ | ElevenLabsSoundEffectsProviderOptions
89
+
90
+ /**
91
+ * ElevenLabs audio generation adapter. Dispatches to music or SFX endpoints
92
+ * based on the model id. Music → `client.music.compose`, SFX →
93
+ * `client.textToSoundEffects.convert`.
94
+ *
95
+ * @example
96
+ * ```ts
97
+ * const music = elevenlabsAudio('music_v1')
98
+ * await generateAudio({ adapter: music, prompt: 'lo-fi beat', duration: 15 })
99
+ *
100
+ * const sfx = elevenlabsAudio('eleven_text_to_sound_v2')
101
+ * await generateAudio({ adapter: sfx, prompt: 'glass shattering', duration: 3 })
102
+ * ```
103
+ */
104
+ export class ElevenLabsAudioAdapter<
105
+ TModel extends ElevenLabsAudioModel,
106
+ > extends BaseAudioAdapter<TModel, ElevenLabsAudioProviderOptions> {
107
+ readonly name = 'elevenlabs' as const
108
+
109
+ private client: ElevenLabsClient
110
+
111
+ constructor(model: TModel, config?: ElevenLabsClientConfig) {
112
+ super(model, config ?? {})
113
+ this.client = createElevenLabsClient(config)
114
+ }
115
+
116
+ async generateAudio(
117
+ options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,
118
+ ): Promise<AudioGenerationResult> {
119
+ const { logger } = options
120
+ logger.request(
121
+ `activity=generateAudio provider=elevenlabs model=${this.model}`,
122
+ { provider: 'elevenlabs', model: this.model },
123
+ )
124
+ try {
125
+ if (isElevenLabsMusicModel(this.model)) {
126
+ return await this.runMusic(options)
127
+ }
128
+ if (isElevenLabsSoundEffectsModel(this.model)) {
129
+ return await this.runSoundEffects(options)
130
+ }
131
+ throw new Error(
132
+ `Unsupported ElevenLabs audio model "${this.model}". Expected one of: music_v1, eleven_text_to_sound_v2, eleven_text_to_sound_v1.`,
133
+ )
134
+ } catch (error) {
135
+ logger.errors('elevenlabs.generateAudio fatal', {
136
+ error,
137
+ source: 'elevenlabs.generateAudio',
138
+ })
139
+ throw error
140
+ }
141
+ }
142
+
143
+ private async runMusic(
144
+ options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,
145
+ ): Promise<AudioGenerationResult> {
146
+ // Gated by isElevenLabsMusicModel() in generateAudio().
147
+ const modelId = this.model as ElevenLabsMusicModel
148
+ const music = (options.modelOptions ?? {}) as ElevenLabsMusicProviderOptions
149
+ const outputFormat = music.outputFormat
150
+
151
+ const stream = await this.client.music.compose({
152
+ modelId,
153
+ ...(options.prompt && !music.compositionPlan
154
+ ? { prompt: options.prompt }
155
+ : {}),
156
+ ...(music.compositionPlan
157
+ ? { compositionPlan: toMusicPrompt(music.compositionPlan) }
158
+ : {}),
159
+ ...(options.duration != null && !music.compositionPlan
160
+ ? { musicLengthMs: Math.round(options.duration * 1000) }
161
+ : {}),
162
+ ...(outputFormat ? { outputFormat } : {}),
163
+ ...(music.seed != null ? { seed: music.seed } : {}),
164
+ ...(music.forceInstrumental != null
165
+ ? { forceInstrumental: music.forceInstrumental }
166
+ : {}),
167
+ ...(music.respectSectionsDurations != null
168
+ ? { respectSectionsDurations: music.respectSectionsDurations }
169
+ : {}),
170
+ })
171
+
172
+ return this.finalize(stream, outputFormat, options.duration)
173
+ }
174
+
175
+ private async runSoundEffects(
176
+ options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,
177
+ ): Promise<AudioGenerationResult> {
178
+ // Gated by isElevenLabsSoundEffectsModel() in generateAudio().
179
+ const modelId = this.model as ElevenLabsSoundEffectsModel
180
+ const sfx = (options.modelOptions ??
181
+ {}) as ElevenLabsSoundEffectsProviderOptions
182
+ const outputFormat = sfx.outputFormat
183
+
184
+ const stream = await this.client.textToSoundEffects.convert({
185
+ text: options.prompt,
186
+ modelId,
187
+ ...(options.duration != null
188
+ ? { durationSeconds: options.duration }
189
+ : {}),
190
+ ...(outputFormat ? { outputFormat } : {}),
191
+ ...(sfx.promptInfluence != null
192
+ ? { promptInfluence: sfx.promptInfluence }
193
+ : {}),
194
+ ...(sfx.loop != null ? { loop: sfx.loop } : {}),
195
+ })
196
+
197
+ return this.finalize(stream, outputFormat, options.duration)
198
+ }
199
+
200
+ private async finalize(
201
+ stream: ReadableStream<Uint8Array>,
202
+ outputFormat: ElevenLabsOutputFormat | undefined,
203
+ duration: number | undefined,
204
+ ): Promise<AudioGenerationResult> {
205
+ const buffer = await readStreamToArrayBuffer(stream)
206
+ const base64 = arrayBufferToBase64(buffer)
207
+ const { contentType } = parseOutputFormat(outputFormat)
208
+ return {
209
+ id: generateId(this.name),
210
+ model: this.model,
211
+ audio: {
212
+ b64Json: base64,
213
+ contentType,
214
+ ...(duration != null ? { duration } : {}),
215
+ },
216
+ }
217
+ }
218
+
219
+ protected override generateId(): string {
220
+ return generateId(this.name)
221
+ }
222
+ }
223
+
224
+ function toMusicPrompt(plan: ElevenLabsMusicCompositionPlan) {
225
+ return {
226
+ positiveGlobalStyles: plan.positiveGlobalStyles ?? [],
227
+ negativeGlobalStyles: plan.negativeGlobalStyles ?? [],
228
+ sections: (plan.sections ?? []).map((section) => ({
229
+ sectionName: section.sectionName,
230
+ positiveLocalStyles: section.positiveLocalStyles ?? [],
231
+ negativeLocalStyles: section.negativeLocalStyles ?? [],
232
+ durationMs: section.durationMs ?? 10000,
233
+ lines: section.lines ?? [],
234
+ })),
235
+ }
236
+ }
237
+
238
+ /**
239
+ * Create an ElevenLabs audio adapter using `ELEVENLABS_API_KEY` from env.
240
+ */
241
+ export function elevenlabsAudio<TModel extends ElevenLabsAudioModel>(
242
+ model: TModel,
243
+ config?: ElevenLabsClientConfig,
244
+ ): ElevenLabsAudioAdapter<TModel> {
245
+ return new ElevenLabsAudioAdapter(model, config)
246
+ }
247
+
248
+ /**
249
+ * Create an ElevenLabs audio adapter with an explicit API key.
250
+ */
251
+ export function createElevenLabsAudio<TModel extends ElevenLabsAudioModel>(
252
+ model: TModel,
253
+ apiKey: string,
254
+ config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
255
+ ): ElevenLabsAudioAdapter<TModel> {
256
+ return new ElevenLabsAudioAdapter(model, { apiKey, ...config })
257
+ }
@@ -0,0 +1,240 @@
1
+ import { BaseTTSAdapter } from '@tanstack/ai/adapters'
2
+ import {
3
+ arrayBufferToBase64,
4
+ createElevenLabsClient,
5
+ generateId,
6
+ parseOutputFormat,
7
+ readStreamToArrayBuffer,
8
+ } from '../utils/client'
9
+ import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
10
+ import type { TTSOptions, TTSResult } from '@tanstack/ai'
11
+ import type { ElevenLabsClientConfig } from '../utils/client'
12
+ import type { ElevenLabsOutputFormat, ElevenLabsTTSModel } from '../model-meta'
13
+
14
+ /**
15
+ * ElevenLabs voice settings overrides. All fields are optional — omitted
16
+ * values fall back to the voice's stored defaults.
17
+ * @see https://elevenlabs.io/docs/api-reference/text-to-speech/convert
18
+ */
19
+ export interface ElevenLabsVoiceSettings {
20
+ /** Voice stability, 0..1. Default 0.5. */
21
+ stability?: number
22
+ /** Similarity boost, 0..1. Default 0.75. */
23
+ similarityBoost?: number
24
+ /** Style exaggeration, 0..1. Default 0. */
25
+ style?: number
26
+ /** Playback speed. Default 1.0. */
27
+ speed?: number
28
+ /** Clarity/presence boost. Default true. */
29
+ useSpeakerBoost?: boolean
30
+ }
31
+
32
+ /**
33
+ * Provider-specific TTS options. `voice` on `generateSpeech()` takes priority
34
+ * over `voiceId` here, but we expose the same field for callers that prefer
35
+ * to keep voice configuration inside the adapter config.
36
+ */
37
+ export interface ElevenLabsSpeechProviderOptions {
38
+ /** ElevenLabs voice ID to synthesize. Required if `generateSpeech().voice` is not set. */
39
+ voiceId?: string
40
+ /** Output audio format encoded as `codec_samplerate[_bitrate]`. Defaults to `mp3_44100_128`. */
41
+ outputFormat?: ElevenLabsOutputFormat
42
+ /** Voice-settings overrides for this request only. */
43
+ voiceSettings?: ElevenLabsVoiceSettings
44
+ /** ISO-639-1 language code to enforce (e.g. `'en'`, `'ja'`). */
45
+ languageCode?: string
46
+ /** Deterministic sampling seed, 0..4294967295. */
47
+ seed?: number
48
+ /** Previous text for stitching adjacent clips. */
49
+ previousText?: string
50
+ /** Next text for stitching adjacent clips. */
51
+ nextText?: string
52
+ /** Previous request IDs for stitching (max 3). */
53
+ previousRequestIds?: Array<string>
54
+ /** Next request IDs for stitching (max 3). */
55
+ nextRequestIds?: Array<string>
56
+ /** Text normalization toggle. Default `'auto'`. */
57
+ applyTextNormalization?: 'auto' | 'on' | 'off'
58
+ /** Language-specific text normalization (currently Japanese only, adds latency). */
59
+ applyLanguageTextNormalization?: boolean
60
+ /** Latency optimization level, 0..4. */
61
+ optimizeStreamingLatency?: number
62
+ /** Enable logging. Set false for zero-retention mode (enterprise only). */
63
+ enableLogging?: boolean
64
+ }
65
+
66
+ /**
67
+ * ElevenLabs text-to-speech adapter built on the official
68
+ * `@elevenlabs/elevenlabs-js` SDK.
69
+ *
70
+ * @example
71
+ * ```ts
72
+ * const adapter = elevenlabsSpeech('eleven_multilingual_v2')
73
+ * const result = await generateSpeech({
74
+ * adapter,
75
+ * text: 'Hello, world!',
76
+ * voice: '21m00Tcm4TlvDq8ikWAM',
77
+ * })
78
+ * ```
79
+ */
80
+ export class ElevenLabsSpeechAdapter<
81
+ TModel extends ElevenLabsTTSModel,
82
+ > extends BaseTTSAdapter<TModel, ElevenLabsSpeechProviderOptions> {
83
+ readonly name = 'elevenlabs' as const
84
+
85
+ private client: ElevenLabsClient
86
+
87
+ constructor(model: TModel, config?: ElevenLabsClientConfig) {
88
+ super(model, config ?? {})
89
+ this.client = createElevenLabsClient(config)
90
+ }
91
+
92
+ async generateSpeech(
93
+ options: TTSOptions<ElevenLabsSpeechProviderOptions>,
94
+ ): Promise<TTSResult> {
95
+ const { logger } = options
96
+ logger.request(
97
+ `activity=generateSpeech provider=elevenlabs model=${this.model}`,
98
+ { provider: 'elevenlabs', model: this.model },
99
+ )
100
+ try {
101
+ const voiceId = options.voice ?? options.modelOptions?.voiceId
102
+ if (!voiceId) {
103
+ throw new Error(
104
+ 'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',
105
+ )
106
+ }
107
+ const {
108
+ outputFormat,
109
+ voiceSettings,
110
+ languageCode,
111
+ seed,
112
+ previousText,
113
+ nextText,
114
+ previousRequestIds,
115
+ nextRequestIds,
116
+ applyTextNormalization,
117
+ applyLanguageTextNormalization,
118
+ optimizeStreamingLatency,
119
+ enableLogging,
120
+ } = options.modelOptions ?? {}
121
+ const effectiveOutputFormat =
122
+ outputFormat ?? inferOutputFormatFromResponseFormat(options.format)
123
+
124
+ const stream = await this.client.textToSpeech.convert(voiceId, {
125
+ text: options.text,
126
+ modelId: this.model,
127
+ ...(effectiveOutputFormat
128
+ ? { outputFormat: effectiveOutputFormat }
129
+ : {}),
130
+ ...(voiceSettings
131
+ ? { voiceSettings: mapVoiceSettings(voiceSettings, options.speed) }
132
+ : options.speed != null
133
+ ? { voiceSettings: { speed: options.speed } }
134
+ : {}),
135
+ ...(languageCode ? { languageCode } : {}),
136
+ ...(seed != null ? { seed } : {}),
137
+ ...(previousText ? { previousText } : {}),
138
+ ...(nextText ? { nextText } : {}),
139
+ ...(previousRequestIds ? { previousRequestIds } : {}),
140
+ ...(nextRequestIds ? { nextRequestIds } : {}),
141
+ ...(applyTextNormalization ? { applyTextNormalization } : {}),
142
+ ...(applyLanguageTextNormalization != null
143
+ ? { applyLanguageTextNormalization }
144
+ : {}),
145
+ ...(optimizeStreamingLatency != null
146
+ ? { optimizeStreamingLatency }
147
+ : {}),
148
+ ...(enableLogging != null ? { enableLogging } : {}),
149
+ })
150
+
151
+ const buffer = await readStreamToArrayBuffer(stream)
152
+ const base64 = arrayBufferToBase64(buffer)
153
+ const { format, contentType } = parseOutputFormat(effectiveOutputFormat)
154
+
155
+ return {
156
+ id: generateId(this.name),
157
+ model: this.model,
158
+ audio: base64,
159
+ format,
160
+ contentType,
161
+ }
162
+ } catch (error) {
163
+ logger.errors('elevenlabs.generateSpeech fatal', {
164
+ error,
165
+ source: 'elevenlabs.generateSpeech',
166
+ })
167
+ throw error
168
+ }
169
+ }
170
+
171
+ protected override generateId(): string {
172
+ return generateId(this.name)
173
+ }
174
+ }
175
+
176
+ function mapVoiceSettings(
177
+ settings: ElevenLabsVoiceSettings,
178
+ speedOverride: number | undefined,
179
+ ): Record<string, unknown> {
180
+ return {
181
+ ...(settings.stability != null ? { stability: settings.stability } : {}),
182
+ ...(settings.similarityBoost != null
183
+ ? { similarityBoost: settings.similarityBoost }
184
+ : {}),
185
+ ...(settings.style != null ? { style: settings.style } : {}),
186
+ ...(speedOverride != null
187
+ ? { speed: speedOverride }
188
+ : settings.speed != null
189
+ ? { speed: settings.speed }
190
+ : {}),
191
+ ...(settings.useSpeakerBoost != null
192
+ ? { useSpeakerBoost: settings.useSpeakerBoost }
193
+ : {}),
194
+ }
195
+ }
196
+
197
+ /**
198
+ * Map the standard TTSOptions `format` (mp3/opus/aac/flac/wav/pcm) to a
199
+ * reasonable ElevenLabs `outputFormat` so callers don't need to know the
200
+ * full codec/samplerate string for the common case.
201
+ */
202
+ function inferOutputFormatFromResponseFormat(
203
+ format: TTSOptions['format'] | undefined,
204
+ ): ElevenLabsOutputFormat | undefined {
205
+ switch (format) {
206
+ case 'mp3':
207
+ return 'mp3_44100_128'
208
+ case 'pcm':
209
+ return 'pcm_44100'
210
+ case 'opus':
211
+ return 'opus_48000_128'
212
+ case undefined:
213
+ return undefined
214
+ default:
215
+ // `aac` / `flac` / `wav` are not native ElevenLabs formats —
216
+ // fall back to mp3 rather than blowing up mid-request.
217
+ return 'mp3_44100_128'
218
+ }
219
+ }
220
+
221
+ /**
222
+ * Create an ElevenLabs speech adapter using `ELEVENLABS_API_KEY` from env.
223
+ */
224
+ export function elevenlabsSpeech<TModel extends ElevenLabsTTSModel>(
225
+ model: TModel,
226
+ config?: ElevenLabsClientConfig,
227
+ ): ElevenLabsSpeechAdapter<TModel> {
228
+ return new ElevenLabsSpeechAdapter(model, config)
229
+ }
230
+
231
+ /**
232
+ * Create an ElevenLabs speech adapter with an explicit API key.
233
+ */
234
+ export function createElevenLabsSpeech<TModel extends ElevenLabsTTSModel>(
235
+ model: TModel,
236
+ apiKey: string,
237
+ config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
238
+ ): ElevenLabsSpeechAdapter<TModel> {
239
+ return new ElevenLabsSpeechAdapter(model, { apiKey, ...config })
240
+ }