@tanstack/ai-elevenlabs 0.1.7 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +93 -0
- package/dist/esm/adapters/audio.js +106 -0
- package/dist/esm/adapters/audio.js.map +1 -0
- package/dist/esm/adapters/speech.d.ts +83 -0
- package/dist/esm/adapters/speech.js +109 -0
- package/dist/esm/adapters/speech.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +79 -0
- package/dist/esm/adapters/transcription.js +142 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/index.d.ts +5 -0
- package/dist/esm/index.js +21 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +42 -0
- package/dist/esm/model-meta.js +32 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/realtime/adapter.d.ts +1 -1
- package/dist/esm/realtime/adapter.js +1 -1
- package/dist/esm/realtime/adapter.js.map +1 -1
- package/dist/esm/realtime/token.d.ts +11 -7
- package/dist/esm/realtime/token.js +8 -32
- package/dist/esm/realtime/token.js.map +1 -1
- package/dist/esm/realtime/types.d.ts +5 -2
- package/dist/esm/utils/client.d.ts +62 -0
- package/dist/esm/utils/client.js +124 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/utils/index.d.ts +1 -0
- package/package.json +15 -7
- package/src/adapters/audio.ts +257 -0
- package/src/adapters/speech.ts +240 -0
- package/src/adapters/transcription.ts +338 -0
- package/src/index.ts +64 -0
- package/src/model-meta.ts +79 -0
- package/src/realtime/adapter.ts +3 -3
- package/src/realtime/token.ts +21 -56
- package/src/realtime/types.ts +5 -2
- package/src/utils/client.ts +196 -0
- package/src/utils/index.ts +10 -0
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import {
|
|
3
|
+
createElevenLabsClient,
|
|
4
|
+
dataUrlToBlob,
|
|
5
|
+
generateId,
|
|
6
|
+
} from '../utils/client'
|
|
7
|
+
import type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'
|
|
8
|
+
import type {
|
|
9
|
+
TranscriptionOptions,
|
|
10
|
+
TranscriptionResult,
|
|
11
|
+
TranscriptionSegment,
|
|
12
|
+
TranscriptionWord,
|
|
13
|
+
} from '@tanstack/ai'
|
|
14
|
+
import type { ElevenLabsClientConfig } from '../utils/client'
|
|
15
|
+
import type { ElevenLabsTranscriptionModel } from '../model-meta'
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Provider-specific options for ElevenLabs Scribe transcription. Fields map
|
|
19
|
+
* 1:1 onto the SDK's `BodySpeechToTextV1SpeechToTextPost` — mirroring the
|
|
20
|
+
* names so documentation stays useful.
|
|
21
|
+
* @see https://elevenlabs.io/docs/api-reference/speech-to-text/convert
|
|
22
|
+
*/
|
|
23
|
+
export interface ElevenLabsTranscriptionProviderOptions {
|
|
24
|
+
/** Annotate non-speech events like (laughter), (footsteps), …. */
|
|
25
|
+
tagAudioEvents?: boolean
|
|
26
|
+
/** Maximum number of speakers in the audio (1..32). */
|
|
27
|
+
numSpeakers?: number
|
|
28
|
+
/** Timestamp granularity for words. */
|
|
29
|
+
timestampsGranularity?: 'word' | 'character' | 'none'
|
|
30
|
+
/** Enable speaker diarization. */
|
|
31
|
+
diarize?: boolean
|
|
32
|
+
/** Diarization threshold (requires `diarize=true` and no `numSpeakers`). */
|
|
33
|
+
diarizationThreshold?: number
|
|
34
|
+
/** Detect speaker roles (agent/customer). Requires diarize=true. */
|
|
35
|
+
detectSpeakerRoles?: boolean
|
|
36
|
+
/** Bias the model towards these keyterms (max 1000). */
|
|
37
|
+
keyterms?: Array<string>
|
|
38
|
+
/**
|
|
39
|
+
* Entity detection: `'all'`, a category (`'pii'`, `'phi'`, `'pci'`,
|
|
40
|
+
* `'other'`, `'offensive_language'`), or a specific entity type.
|
|
41
|
+
*/
|
|
42
|
+
entityDetection?: string
|
|
43
|
+
/** Redact entities from the transcript text. Must be a subset of `entityDetection`. */
|
|
44
|
+
entityRedaction?: string
|
|
45
|
+
/** How redacted entities are formatted. */
|
|
46
|
+
entityRedactionMode?: string
|
|
47
|
+
/** Whether to skip filler words / non-speech sounds (scribe_v2 only). */
|
|
48
|
+
noVerbatim?: boolean
|
|
49
|
+
/** Sampling temperature (0..2). */
|
|
50
|
+
temperature?: number
|
|
51
|
+
/** Deterministic sampling seed (0..2147483647). */
|
|
52
|
+
seed?: number
|
|
53
|
+
/** Use `false` for zero-retention mode (enterprise only). */
|
|
54
|
+
enableLogging?: boolean
|
|
55
|
+
/** Multi-channel audio with one speaker per channel. Max 5 channels. */
|
|
56
|
+
useMultiChannel?: boolean
|
|
57
|
+
/**
|
|
58
|
+
* Hint for audio format. Use `'pcm_s16le_16'` to skip encoding for 16-bit
|
|
59
|
+
* PCM @ 16kHz mono little-endian inputs (lower latency).
|
|
60
|
+
*/
|
|
61
|
+
fileFormat?: 'pcm_s16le_16' | 'other'
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* ElevenLabs speech-to-text adapter built on the official SDK's Scribe family.
|
|
66
|
+
*
|
|
67
|
+
* @example
|
|
68
|
+
* ```ts
|
|
69
|
+
* const adapter = elevenlabsTranscription('scribe_v1')
|
|
70
|
+
* const result = await generateTranscription({
|
|
71
|
+
* adapter,
|
|
72
|
+
* audio: fileInput,
|
|
73
|
+
* language: 'en',
|
|
74
|
+
* })
|
|
75
|
+
* ```
|
|
76
|
+
*/
|
|
77
|
+
export class ElevenLabsTranscriptionAdapter<
|
|
78
|
+
TModel extends ElevenLabsTranscriptionModel,
|
|
79
|
+
> extends BaseTranscriptionAdapter<
|
|
80
|
+
TModel,
|
|
81
|
+
ElevenLabsTranscriptionProviderOptions
|
|
82
|
+
> {
|
|
83
|
+
readonly name = 'elevenlabs' as const
|
|
84
|
+
|
|
85
|
+
private client: ElevenLabsClient
|
|
86
|
+
|
|
87
|
+
constructor(model: TModel, config?: ElevenLabsClientConfig) {
|
|
88
|
+
super(model, config ?? {})
|
|
89
|
+
this.client = createElevenLabsClient(config)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
async transcribe(
|
|
93
|
+
options: TranscriptionOptions<ElevenLabsTranscriptionProviderOptions>,
|
|
94
|
+
): Promise<TranscriptionResult> {
|
|
95
|
+
const { logger } = options
|
|
96
|
+
logger.request(
|
|
97
|
+
`activity=generateTranscription provider=elevenlabs model=${this.model}`,
|
|
98
|
+
{ provider: 'elevenlabs', model: this.model },
|
|
99
|
+
)
|
|
100
|
+
try {
|
|
101
|
+
const modelOpts = options.modelOptions ?? {}
|
|
102
|
+
const audioInput = normalizeAudioInput(options.audio)
|
|
103
|
+
|
|
104
|
+
const response = await this.client.speechToText.convert({
|
|
105
|
+
modelId: this.model,
|
|
106
|
+
...(audioInput.kind === 'file'
|
|
107
|
+
? { file: audioInput.value }
|
|
108
|
+
: { cloudStorageUrl: audioInput.value }),
|
|
109
|
+
...(options.language ? { languageCode: options.language } : {}),
|
|
110
|
+
...(modelOpts.tagAudioEvents != null
|
|
111
|
+
? { tagAudioEvents: modelOpts.tagAudioEvents }
|
|
112
|
+
: {}),
|
|
113
|
+
...(modelOpts.numSpeakers != null
|
|
114
|
+
? { numSpeakers: modelOpts.numSpeakers }
|
|
115
|
+
: {}),
|
|
116
|
+
...(modelOpts.timestampsGranularity
|
|
117
|
+
? { timestampsGranularity: modelOpts.timestampsGranularity }
|
|
118
|
+
: {}),
|
|
119
|
+
...(modelOpts.diarize != null ? { diarize: modelOpts.diarize } : {}),
|
|
120
|
+
...(modelOpts.diarizationThreshold != null
|
|
121
|
+
? { diarizationThreshold: modelOpts.diarizationThreshold }
|
|
122
|
+
: {}),
|
|
123
|
+
...(modelOpts.detectSpeakerRoles != null
|
|
124
|
+
? { detectSpeakerRoles: modelOpts.detectSpeakerRoles }
|
|
125
|
+
: {}),
|
|
126
|
+
...(modelOpts.keyterms ? { keyterms: modelOpts.keyterms } : {}),
|
|
127
|
+
...(modelOpts.entityDetection
|
|
128
|
+
? { entityDetection: modelOpts.entityDetection }
|
|
129
|
+
: {}),
|
|
130
|
+
...(modelOpts.entityRedaction
|
|
131
|
+
? { entityRedaction: modelOpts.entityRedaction }
|
|
132
|
+
: {}),
|
|
133
|
+
...(modelOpts.entityRedactionMode
|
|
134
|
+
? { entityRedactionMode: modelOpts.entityRedactionMode }
|
|
135
|
+
: {}),
|
|
136
|
+
...(modelOpts.noVerbatim != null
|
|
137
|
+
? { noVerbatim: modelOpts.noVerbatim }
|
|
138
|
+
: {}),
|
|
139
|
+
...(modelOpts.temperature != null
|
|
140
|
+
? { temperature: modelOpts.temperature }
|
|
141
|
+
: {}),
|
|
142
|
+
...(modelOpts.seed != null ? { seed: modelOpts.seed } : {}),
|
|
143
|
+
...(modelOpts.enableLogging != null
|
|
144
|
+
? { enableLogging: modelOpts.enableLogging }
|
|
145
|
+
: {}),
|
|
146
|
+
...(modelOpts.useMultiChannel != null
|
|
147
|
+
? { useMultiChannel: modelOpts.useMultiChannel }
|
|
148
|
+
: {}),
|
|
149
|
+
...(modelOpts.fileFormat ? { fileFormat: modelOpts.fileFormat } : {}),
|
|
150
|
+
} as Parameters<ElevenLabsClient['speechToText']['convert']>[0])
|
|
151
|
+
|
|
152
|
+
return this.transformResponse(response)
|
|
153
|
+
} catch (error) {
|
|
154
|
+
logger.errors('elevenlabs.generateTranscription fatal', {
|
|
155
|
+
error,
|
|
156
|
+
source: 'elevenlabs.generateTranscription',
|
|
157
|
+
})
|
|
158
|
+
throw error
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
private transformResponse(
|
|
163
|
+
response: Awaited<ReturnType<ElevenLabsClient['speechToText']['convert']>>,
|
|
164
|
+
): TranscriptionResult {
|
|
165
|
+
// The SDK types this as a union of single- and multi-channel responses.
|
|
166
|
+
// We treat multi-channel as "join the channel transcripts" — consumers
|
|
167
|
+
// who care about per-channel detail can re-parse from `modelOptions`.
|
|
168
|
+
const data = response as unknown as {
|
|
169
|
+
text?: string
|
|
170
|
+
languageCode?: string
|
|
171
|
+
languageProbability?: number
|
|
172
|
+
words?: Array<{
|
|
173
|
+
text: string
|
|
174
|
+
start?: number
|
|
175
|
+
end?: number
|
|
176
|
+
type: string
|
|
177
|
+
speakerId?: string
|
|
178
|
+
}>
|
|
179
|
+
audioDurationSecs?: number
|
|
180
|
+
transcripts?: Array<{
|
|
181
|
+
text?: string
|
|
182
|
+
languageCode?: string
|
|
183
|
+
words?: Array<{
|
|
184
|
+
text: string
|
|
185
|
+
start?: number
|
|
186
|
+
end?: number
|
|
187
|
+
type: string
|
|
188
|
+
speakerId?: string
|
|
189
|
+
}>
|
|
190
|
+
audioDurationSecs?: number
|
|
191
|
+
}>
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
if (data.transcripts) {
|
|
195
|
+
const joinedText = data.transcripts
|
|
196
|
+
.map((t) => t.text ?? '')
|
|
197
|
+
.filter(Boolean)
|
|
198
|
+
.join('\n')
|
|
199
|
+
const joinedWords = data.transcripts.flatMap((t) => t.words ?? [])
|
|
200
|
+
const duration = data.transcripts.reduce(
|
|
201
|
+
(max, t) => Math.max(max, t.audioDurationSecs ?? 0),
|
|
202
|
+
0,
|
|
203
|
+
)
|
|
204
|
+
const firstLang = data.transcripts.find(
|
|
205
|
+
(t) => t.languageCode,
|
|
206
|
+
)?.languageCode
|
|
207
|
+
return {
|
|
208
|
+
id: generateId(this.name),
|
|
209
|
+
model: this.model,
|
|
210
|
+
text: joinedText,
|
|
211
|
+
...(firstLang ? { language: firstLang } : {}),
|
|
212
|
+
...(duration ? { duration } : {}),
|
|
213
|
+
...buildWordsAndSegments(joinedWords),
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
return {
|
|
218
|
+
id: generateId(this.name),
|
|
219
|
+
model: this.model,
|
|
220
|
+
text: data.text ?? '',
|
|
221
|
+
...(data.languageCode ? { language: data.languageCode } : {}),
|
|
222
|
+
...(data.audioDurationSecs ? { duration: data.audioDurationSecs } : {}),
|
|
223
|
+
...buildWordsAndSegments(data.words ?? []),
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
protected override generateId(): string {
|
|
228
|
+
return generateId(this.name)
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
type NormalizedAudio =
|
|
233
|
+
| { kind: 'file'; value: Blob }
|
|
234
|
+
| { kind: 'url'; value: string }
|
|
235
|
+
|
|
236
|
+
function normalizeAudioInput(
|
|
237
|
+
audio: TranscriptionOptions['audio'],
|
|
238
|
+
): NormalizedAudio {
|
|
239
|
+
if (audio instanceof ArrayBuffer) {
|
|
240
|
+
return { kind: 'file', value: new Blob([audio]) }
|
|
241
|
+
}
|
|
242
|
+
if (typeof audio === 'string') {
|
|
243
|
+
const blob = dataUrlToBlob(audio)
|
|
244
|
+
if (blob) return { kind: 'file', value: blob }
|
|
245
|
+
return { kind: 'url', value: audio }
|
|
246
|
+
}
|
|
247
|
+
// Blob or File both fit the SDK's `Uploadable` contract.
|
|
248
|
+
return { kind: 'file', value: audio }
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
function buildWordsAndSegments(
|
|
252
|
+
words: Array<{
|
|
253
|
+
text: string
|
|
254
|
+
start?: number
|
|
255
|
+
end?: number
|
|
256
|
+
type: string
|
|
257
|
+
speakerId?: string
|
|
258
|
+
}>,
|
|
259
|
+
): {
|
|
260
|
+
words?: Array<TranscriptionWord>
|
|
261
|
+
segments?: Array<TranscriptionSegment>
|
|
262
|
+
} {
|
|
263
|
+
const timedWords = words.filter(
|
|
264
|
+
(w) =>
|
|
265
|
+
typeof w.start === 'number' &&
|
|
266
|
+
typeof w.end === 'number' &&
|
|
267
|
+
w.type !== 'spacing',
|
|
268
|
+
)
|
|
269
|
+
if (timedWords.length === 0) return {}
|
|
270
|
+
|
|
271
|
+
const outWords: Array<TranscriptionWord> = timedWords.map((w) => ({
|
|
272
|
+
word: w.text,
|
|
273
|
+
start: w.start!,
|
|
274
|
+
end: w.end!,
|
|
275
|
+
}))
|
|
276
|
+
|
|
277
|
+
// Group contiguous words that share a speaker into segments. If no speaker
|
|
278
|
+
// is ever set, we still emit one segment per sentence-ish grouping.
|
|
279
|
+
const segments: Array<TranscriptionSegment> = []
|
|
280
|
+
let current: {
|
|
281
|
+
start: number
|
|
282
|
+
end: number
|
|
283
|
+
text: string
|
|
284
|
+
speaker?: string
|
|
285
|
+
} | null = null
|
|
286
|
+
|
|
287
|
+
for (const w of timedWords) {
|
|
288
|
+
if (!current) {
|
|
289
|
+
current = {
|
|
290
|
+
start: w.start!,
|
|
291
|
+
end: w.end!,
|
|
292
|
+
text: w.text,
|
|
293
|
+
...(w.speakerId ? { speaker: w.speakerId } : {}),
|
|
294
|
+
}
|
|
295
|
+
continue
|
|
296
|
+
}
|
|
297
|
+
if (w.speakerId && current.speaker !== w.speakerId) {
|
|
298
|
+
segments.push({ id: segments.length, ...current })
|
|
299
|
+
current = {
|
|
300
|
+
start: w.start!,
|
|
301
|
+
end: w.end!,
|
|
302
|
+
text: w.text,
|
|
303
|
+
speaker: w.speakerId,
|
|
304
|
+
}
|
|
305
|
+
continue
|
|
306
|
+
}
|
|
307
|
+
current.end = w.end!
|
|
308
|
+
current.text = current.text ? `${current.text} ${w.text}` : w.text
|
|
309
|
+
}
|
|
310
|
+
if (current) segments.push({ id: segments.length, ...current })
|
|
311
|
+
|
|
312
|
+
return { words: outWords, segments }
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/**
|
|
316
|
+
* Create an ElevenLabs transcription adapter using `ELEVENLABS_API_KEY` from env.
|
|
317
|
+
*/
|
|
318
|
+
export function elevenlabsTranscription<
|
|
319
|
+
TModel extends ElevenLabsTranscriptionModel,
|
|
320
|
+
>(
|
|
321
|
+
model: TModel,
|
|
322
|
+
config?: ElevenLabsClientConfig,
|
|
323
|
+
): ElevenLabsTranscriptionAdapter<TModel> {
|
|
324
|
+
return new ElevenLabsTranscriptionAdapter(model, config)
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* Create an ElevenLabs transcription adapter with an explicit API key.
|
|
329
|
+
*/
|
|
330
|
+
export function createElevenLabsTranscription<
|
|
331
|
+
TModel extends ElevenLabsTranscriptionModel,
|
|
332
|
+
>(
|
|
333
|
+
model: TModel,
|
|
334
|
+
apiKey: string,
|
|
335
|
+
config?: Omit<ElevenLabsClientConfig, 'apiKey'>,
|
|
336
|
+
): ElevenLabsTranscriptionAdapter<TModel> {
|
|
337
|
+
return new ElevenLabsTranscriptionAdapter(model, { apiKey, ...config })
|
|
338
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -11,3 +11,67 @@ export type {
|
|
|
11
11
|
ElevenLabsVADConfig,
|
|
12
12
|
ElevenLabsClientTool,
|
|
13
13
|
} from './realtime/index'
|
|
14
|
+
|
|
15
|
+
// ============================================================================
|
|
16
|
+
// Speech (Text-to-Speech) Adapter
|
|
17
|
+
// ============================================================================
|
|
18
|
+
|
|
19
|
+
export {
|
|
20
|
+
ElevenLabsSpeechAdapter,
|
|
21
|
+
createElevenLabsSpeech,
|
|
22
|
+
elevenlabsSpeech,
|
|
23
|
+
type ElevenLabsSpeechProviderOptions,
|
|
24
|
+
type ElevenLabsVoiceSettings,
|
|
25
|
+
} from './adapters/speech'
|
|
26
|
+
|
|
27
|
+
// ============================================================================
|
|
28
|
+
// Audio (Music + Sound Effects) Adapter
|
|
29
|
+
// ============================================================================
|
|
30
|
+
|
|
31
|
+
export {
|
|
32
|
+
ElevenLabsAudioAdapter,
|
|
33
|
+
createElevenLabsAudio,
|
|
34
|
+
elevenlabsAudio,
|
|
35
|
+
type ElevenLabsAudioProviderOptions,
|
|
36
|
+
type ElevenLabsMusicProviderOptions,
|
|
37
|
+
type ElevenLabsSoundEffectsProviderOptions,
|
|
38
|
+
type ElevenLabsMusicCompositionPlan,
|
|
39
|
+
} from './adapters/audio'
|
|
40
|
+
|
|
41
|
+
// ============================================================================
|
|
42
|
+
// Transcription (Speech-to-Text) Adapter
|
|
43
|
+
// ============================================================================
|
|
44
|
+
|
|
45
|
+
export {
|
|
46
|
+
ElevenLabsTranscriptionAdapter,
|
|
47
|
+
createElevenLabsTranscription,
|
|
48
|
+
elevenlabsTranscription,
|
|
49
|
+
type ElevenLabsTranscriptionProviderOptions,
|
|
50
|
+
} from './adapters/transcription'
|
|
51
|
+
|
|
52
|
+
// ============================================================================
|
|
53
|
+
// Model Metadata
|
|
54
|
+
// ============================================================================
|
|
55
|
+
|
|
56
|
+
export {
|
|
57
|
+
ELEVENLABS_TTS_MODELS,
|
|
58
|
+
ELEVENLABS_AUDIO_MODELS,
|
|
59
|
+
ELEVENLABS_TRANSCRIPTION_MODELS,
|
|
60
|
+
isElevenLabsMusicModel,
|
|
61
|
+
isElevenLabsSoundEffectsModel,
|
|
62
|
+
type ElevenLabsTTSModel,
|
|
63
|
+
type ElevenLabsAudioModel,
|
|
64
|
+
type ElevenLabsMusicModel,
|
|
65
|
+
type ElevenLabsSoundEffectsModel,
|
|
66
|
+
type ElevenLabsTranscriptionModel,
|
|
67
|
+
type ElevenLabsOutputFormat,
|
|
68
|
+
} from './model-meta'
|
|
69
|
+
|
|
70
|
+
// ============================================================================
|
|
71
|
+
// Utilities
|
|
72
|
+
// ============================================================================
|
|
73
|
+
|
|
74
|
+
export {
|
|
75
|
+
getElevenLabsApiKeyFromEnv,
|
|
76
|
+
type ElevenLabsClientConfig,
|
|
77
|
+
} from './utils/index'
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import type { ElevenLabs } from '@elevenlabs/elevenlabs-js'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* ElevenLabs model identifiers. The lists below are the source of truth —
|
|
5
|
+
* callers are blocked from passing unknown model IDs. Keep them in sync with
|
|
6
|
+
* the ElevenLabs SDK via the automated update pipeline.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Text-to-speech models.
|
|
11
|
+
* @see https://elevenlabs.io/docs/models
|
|
12
|
+
*/
|
|
13
|
+
export const ELEVENLABS_TTS_MODELS = [
|
|
14
|
+
'eleven_v3',
|
|
15
|
+
'eleven_multilingual_v2',
|
|
16
|
+
'eleven_flash_v2_5',
|
|
17
|
+
'eleven_flash_v2',
|
|
18
|
+
'eleven_turbo_v2_5',
|
|
19
|
+
'eleven_turbo_v2',
|
|
20
|
+
'eleven_monolingual_v1',
|
|
21
|
+
] as const
|
|
22
|
+
|
|
23
|
+
export type ElevenLabsTTSModel = (typeof ELEVENLABS_TTS_MODELS)[number]
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Audio generation models — music (`music_v1`) + sound effects
|
|
27
|
+
* (`eleven_text_to_sound_v*`) share one `generateAudio` adapter.
|
|
28
|
+
* The adapter dispatches by model id so callers pick behavior via the model.
|
|
29
|
+
*
|
|
30
|
+
* @see https://elevenlabs.io/docs/overview/capabilities/music
|
|
31
|
+
* @see https://elevenlabs.io/docs/overview/capabilities/sound-effects
|
|
32
|
+
*/
|
|
33
|
+
export const ELEVENLABS_AUDIO_MODELS = [
|
|
34
|
+
'music_v1',
|
|
35
|
+
'eleven_text_to_sound_v2',
|
|
36
|
+
'eleven_text_to_sound_v1',
|
|
37
|
+
] as const
|
|
38
|
+
|
|
39
|
+
export type ElevenLabsAudioModel = (typeof ELEVENLABS_AUDIO_MODELS)[number]
|
|
40
|
+
|
|
41
|
+
/** Music models within the audio family. */
|
|
42
|
+
export type ElevenLabsMusicModel = 'music_v1'
|
|
43
|
+
/** SFX models within the audio family. */
|
|
44
|
+
export type ElevenLabsSoundEffectsModel =
|
|
45
|
+
| 'eleven_text_to_sound_v2'
|
|
46
|
+
| 'eleven_text_to_sound_v1'
|
|
47
|
+
|
|
48
|
+
export function isElevenLabsMusicModel(
|
|
49
|
+
model: string,
|
|
50
|
+
): model is ElevenLabsMusicModel {
|
|
51
|
+
return model === 'music_v1'
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function isElevenLabsSoundEffectsModel(
|
|
55
|
+
model: string,
|
|
56
|
+
): model is ElevenLabsSoundEffectsModel {
|
|
57
|
+
return model.startsWith('eleven_text_to_sound_')
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Speech-to-text (transcription) models — Scribe family.
|
|
62
|
+
* @see https://elevenlabs.io/docs/overview/capabilities/speech-to-text
|
|
63
|
+
*/
|
|
64
|
+
export const ELEVENLABS_TRANSCRIPTION_MODELS = [
|
|
65
|
+
'scribe_v2',
|
|
66
|
+
'scribe_v1',
|
|
67
|
+
] as const
|
|
68
|
+
|
|
69
|
+
export type ElevenLabsTranscriptionModel =
|
|
70
|
+
(typeof ELEVENLABS_TRANSCRIPTION_MODELS)[number]
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Supported `output_format` strings, encoded as `codec_samplerate[_bitrate]`.
|
|
74
|
+
* Aliased to the SDK's `AllowedOutputFormats` so the list stays in sync
|
|
75
|
+
* automatically whenever the `@elevenlabs/elevenlabs-js` dependency is bumped.
|
|
76
|
+
*
|
|
77
|
+
* @see https://elevenlabs.io/docs/api-reference/text-to-speech/convert
|
|
78
|
+
*/
|
|
79
|
+
export type ElevenLabsOutputFormat = ElevenLabs.AllowedOutputFormats
|
package/src/realtime/adapter.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { Conversation } from '@
|
|
1
|
+
import { Conversation } from '@elevenlabs/client'
|
|
2
2
|
import { resolveDebugOption } from '@tanstack/ai/adapter-internals'
|
|
3
3
|
import type {
|
|
4
4
|
AnyClientTool,
|
|
@@ -18,7 +18,7 @@ import type { ElevenLabsRealtimeOptions } from './types'
|
|
|
18
18
|
/**
|
|
19
19
|
* Creates an ElevenLabs realtime adapter for client-side use.
|
|
20
20
|
*
|
|
21
|
-
* Wraps the @
|
|
21
|
+
* Wraps the @elevenlabs/client SDK for voice conversations.
|
|
22
22
|
*
|
|
23
23
|
* @param options - Optional configuration
|
|
24
24
|
* @returns A RealtimeAdapter for use with RealtimeClient
|
|
@@ -91,7 +91,7 @@ async function createElevenLabsConnection(
|
|
|
91
91
|
}
|
|
92
92
|
|
|
93
93
|
// Convert TanStack tool definitions to ElevenLabs clientTools format.
|
|
94
|
-
// @
|
|
94
|
+
// @elevenlabs/client expects plain async functions, not objects.
|
|
95
95
|
const elevenLabsClientTools: Record<
|
|
96
96
|
string,
|
|
97
97
|
(params: unknown) => Promise<string>
|
package/src/realtime/token.ts
CHANGED
|
@@ -1,40 +1,19 @@
|
|
|
1
|
+
import {
|
|
2
|
+
createElevenLabsClient,
|
|
3
|
+
getElevenLabsAgentIdFromEnv,
|
|
4
|
+
} from '../utils/client'
|
|
1
5
|
import type { RealtimeToken, RealtimeTokenAdapter } from '@tanstack/ai'
|
|
2
6
|
import type { ElevenLabsRealtimeTokenOptions } from './types'
|
|
3
7
|
|
|
4
|
-
const ELEVENLABS_API_URL = 'https://api.elevenlabs.io/v1'
|
|
5
|
-
|
|
6
|
-
/**
|
|
7
|
-
* Get ElevenLabs API key from environment
|
|
8
|
-
*/
|
|
9
|
-
function getElevenLabsApiKey(): string {
|
|
10
|
-
// Check process.env (Node.js)
|
|
11
|
-
if (typeof process !== 'undefined' && process.env.ELEVENLABS_API_KEY) {
|
|
12
|
-
return process.env.ELEVENLABS_API_KEY
|
|
13
|
-
}
|
|
14
|
-
|
|
15
|
-
// Check window.env (Browser with injected env)
|
|
16
|
-
if (
|
|
17
|
-
typeof window !== 'undefined' &&
|
|
18
|
-
(window as unknown as { env?: { ELEVENLABS_API_KEY?: string } }).env
|
|
19
|
-
?.ELEVENLABS_API_KEY
|
|
20
|
-
) {
|
|
21
|
-
return (window as unknown as { env: { ELEVENLABS_API_KEY: string } }).env
|
|
22
|
-
.ELEVENLABS_API_KEY
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
throw new Error(
|
|
26
|
-
'ELEVENLABS_API_KEY not found in environment variables. ' +
|
|
27
|
-
'Please set ELEVENLABS_API_KEY in your environment.',
|
|
28
|
-
)
|
|
29
|
-
}
|
|
30
|
-
|
|
31
8
|
/**
|
|
32
9
|
* Creates an ElevenLabs realtime token adapter.
|
|
33
10
|
*
|
|
34
|
-
*
|
|
35
|
-
* The signed URL is valid for
|
|
11
|
+
* Uses the official `@elevenlabs/elevenlabs-js` SDK to request a signed URL
|
|
12
|
+
* for client-side conversation connections. The signed URL is valid for
|
|
13
|
+
* 30 minutes.
|
|
36
14
|
*
|
|
37
|
-
* @param options - Configuration
|
|
15
|
+
* @param options - Configuration. `agentId` falls back to
|
|
16
|
+
* `ELEVENLABS_AGENT_ID` in the environment when omitted.
|
|
38
17
|
* @returns A RealtimeTokenAdapter for use with realtimeToken()
|
|
39
18
|
*
|
|
40
19
|
* @example
|
|
@@ -42,51 +21,37 @@ function getElevenLabsApiKey(): string {
|
|
|
42
21
|
* import { realtimeToken } from '@tanstack/ai'
|
|
43
22
|
* import { elevenlabsRealtimeToken } from '@tanstack/ai-elevenlabs'
|
|
44
23
|
*
|
|
24
|
+
* // Reads ELEVENLABS_AGENT_ID from env:
|
|
25
|
+
* const token = await realtimeToken({ adapter: elevenlabsRealtimeToken() })
|
|
26
|
+
*
|
|
27
|
+
* // Or pass explicitly:
|
|
45
28
|
* const token = await realtimeToken({
|
|
46
|
-
* adapter: elevenlabsRealtimeToken({
|
|
47
|
-
* agentId: 'your-agent-id',
|
|
48
|
-
* }),
|
|
29
|
+
* adapter: elevenlabsRealtimeToken({ agentId: 'your-agent-id' }),
|
|
49
30
|
* })
|
|
50
31
|
* ```
|
|
51
32
|
*/
|
|
52
33
|
export function elevenlabsRealtimeToken(
|
|
53
|
-
options: ElevenLabsRealtimeTokenOptions,
|
|
34
|
+
options: ElevenLabsRealtimeTokenOptions = {},
|
|
54
35
|
): RealtimeTokenAdapter {
|
|
55
|
-
const
|
|
36
|
+
const client = createElevenLabsClient()
|
|
56
37
|
|
|
57
38
|
return {
|
|
58
39
|
provider: 'elevenlabs',
|
|
59
40
|
|
|
60
41
|
async generateToken(): Promise<RealtimeToken> {
|
|
61
|
-
const {
|
|
42
|
+
const { overrides } = options
|
|
43
|
+
const agentId = options.agentId ?? getElevenLabsAgentIdFromEnv()
|
|
62
44
|
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
`${ELEVENLABS_API_URL}/convai/conversation/get_signed_url?agent_id=${agentId}`,
|
|
66
|
-
{
|
|
67
|
-
method: 'GET',
|
|
68
|
-
headers: {
|
|
69
|
-
'xi-api-key': apiKey,
|
|
70
|
-
},
|
|
71
|
-
},
|
|
45
|
+
const response = await client.conversationalAi.conversations.getSignedUrl(
|
|
46
|
+
{ agentId },
|
|
72
47
|
)
|
|
73
48
|
|
|
74
|
-
if (!response.ok) {
|
|
75
|
-
const errorText = await response.text()
|
|
76
|
-
throw new Error(
|
|
77
|
-
`ElevenLabs signed URL request failed: ${response.status} ${errorText}`,
|
|
78
|
-
)
|
|
79
|
-
}
|
|
80
|
-
|
|
81
|
-
const data = await response.json()
|
|
82
|
-
const signedUrl = data.signed_url as string
|
|
83
|
-
|
|
84
49
|
// Signed URLs are valid for 30 minutes
|
|
85
50
|
const expiresAt = Date.now() + 30 * 60 * 1000
|
|
86
51
|
|
|
87
52
|
return {
|
|
88
53
|
provider: 'elevenlabs',
|
|
89
|
-
token: signedUrl,
|
|
54
|
+
token: response.signedUrl,
|
|
90
55
|
expiresAt,
|
|
91
56
|
config: {
|
|
92
57
|
voice: overrides?.voiceId,
|
package/src/realtime/types.ts
CHANGED
|
@@ -4,8 +4,11 @@ import type { DebugOption } from '@tanstack/ai'
|
|
|
4
4
|
* Options for the ElevenLabs realtime token adapter
|
|
5
5
|
*/
|
|
6
6
|
export interface ElevenLabsRealtimeTokenOptions {
|
|
7
|
-
/**
|
|
8
|
-
|
|
7
|
+
/**
|
|
8
|
+
* Agent ID configured in ElevenLabs dashboard. Falls back to
|
|
9
|
+
* `ELEVENLABS_AGENT_ID` in the environment when omitted.
|
|
10
|
+
*/
|
|
11
|
+
agentId?: string
|
|
9
12
|
/** Optional override values for the agent */
|
|
10
13
|
overrides?: {
|
|
11
14
|
/** Custom voice ID to use */
|