@tanstack/ai-elevenlabs 0.2.26 → 0.2.28
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +3 -5
- package/src/realtime/adapter.ts +2 -1
- package/dist/esm/adapters/audio.d.ts +0 -93
- package/dist/esm/adapters/audio.js +0 -107
- package/dist/esm/adapters/audio.js.map +0 -1
- package/dist/esm/adapters/speech.d.ts +0 -83
- package/dist/esm/adapters/speech.js +0 -113
- package/dist/esm/adapters/speech.js.map +0 -1
- package/dist/esm/adapters/transcription.d.ts +0 -79
- package/dist/esm/adapters/transcription.js +0 -143
- package/dist/esm/adapters/transcription.js.map +0 -1
- package/dist/esm/index.d.ts +0 -7
- package/dist/esm/index.js +0 -27
- package/dist/esm/index.js.map +0 -1
- package/dist/esm/model-meta.d.ts +0 -42
- package/dist/esm/model-meta.js +0 -32
- package/dist/esm/model-meta.js.map +0 -1
- package/dist/esm/realtime/adapter.d.ts +0 -22
- package/dist/esm/realtime/adapter.js +0 -218
- package/dist/esm/realtime/adapter.js.map +0 -1
- package/dist/esm/realtime/index.d.ts +0 -3
- package/dist/esm/realtime/token.d.ts +0 -28
- package/dist/esm/realtime/token.js +0 -39
- package/dist/esm/realtime/token.js.map +0 -1
- package/dist/esm/realtime/types.d.ts +0 -60
- package/dist/esm/utils/client.d.ts +0 -63
- package/dist/esm/utils/client.js +0 -118
- package/dist/esm/utils/client.js.map +0 -1
- package/dist/esm/utils/index.d.ts +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai-elevenlabs",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.28",
|
|
4
4
|
"description": "ElevenLabs adapter for TanStack AI realtime voice, text-to-speech, transcription, music, and sound effects.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -53,13 +53,11 @@
|
|
|
53
53
|
"@tanstack/ai-utils": "0.3.0"
|
|
54
54
|
},
|
|
55
55
|
"peerDependencies": {
|
|
56
|
-
"@tanstack/ai": "^0.
|
|
57
|
-
"@tanstack/ai-client": "^0.18.2"
|
|
56
|
+
"@tanstack/ai": "^0.36.0"
|
|
58
57
|
},
|
|
59
58
|
"devDependencies": {
|
|
60
59
|
"@vitest/coverage-v8": "4.0.14",
|
|
61
|
-
"@tanstack/ai": "0.
|
|
62
|
-
"@tanstack/ai-client": "0.18.2"
|
|
60
|
+
"@tanstack/ai": "0.36.0"
|
|
63
61
|
},
|
|
64
62
|
"scripts": {
|
|
65
63
|
"build": "vite build",
|
package/src/realtime/adapter.ts
CHANGED
|
@@ -3,6 +3,8 @@ import { resolveDebugOption } from '@tanstack/ai/adapter-internals'
|
|
|
3
3
|
import type {
|
|
4
4
|
AnyClientTool,
|
|
5
5
|
AudioVisualization,
|
|
6
|
+
RealtimeAdapter,
|
|
7
|
+
RealtimeConnection,
|
|
6
8
|
RealtimeEvent,
|
|
7
9
|
RealtimeEventHandler,
|
|
8
10
|
RealtimeMessage,
|
|
@@ -11,7 +13,6 @@ import type {
|
|
|
11
13
|
RealtimeToken,
|
|
12
14
|
} from '@tanstack/ai'
|
|
13
15
|
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
14
|
-
import type { RealtimeAdapter, RealtimeConnection } from '@tanstack/ai-client'
|
|
15
16
|
import type { ElevenLabsRealtimeOptions } from './types'
|
|
16
17
|
|
|
17
18
|
/**
|
|
@@ -1,93 +0,0 @@
|
|
|
1
|
-
import { BaseAudioAdapter } from '@tanstack/ai/adapters';
|
|
2
|
-
import { AudioGenerationOptions, AudioGenerationResult } from '@tanstack/ai';
|
|
3
|
-
import { ElevenLabsClientConfig } from '../utils/client.js';
|
|
4
|
-
import { ElevenLabsAudioModel, ElevenLabsOutputFormat } from '../model-meta.js';
|
|
5
|
-
/**
|
|
6
|
-
* Structured composition plan for ElevenLabs music generation. Mutually
|
|
7
|
-
* exclusive with a free-form `prompt` on the `generateAudio()` call — when
|
|
8
|
-
* supplied, `prompt` is ignored by ElevenLabs.
|
|
9
|
-
*
|
|
10
|
-
* We mirror the SDK's camelCase naming. Lengths are in milliseconds.
|
|
11
|
-
* @see https://elevenlabs.io/docs/api-reference/music/compose
|
|
12
|
-
*/
|
|
13
|
-
export interface ElevenLabsMusicCompositionPlan {
|
|
14
|
-
/** Positive global style descriptors (mood, instruments, tempo, …). */
|
|
15
|
-
positiveGlobalStyles?: Array<string>;
|
|
16
|
-
/** Negative global style descriptors — styles to avoid. */
|
|
17
|
-
negativeGlobalStyles?: Array<string>;
|
|
18
|
-
/** Section definitions (verse/chorus/bridge/…) with local style hints. */
|
|
19
|
-
sections?: Array<{
|
|
20
|
-
sectionName: string;
|
|
21
|
-
positiveLocalStyles?: Array<string>;
|
|
22
|
-
negativeLocalStyles?: Array<string>;
|
|
23
|
-
durationMs?: number;
|
|
24
|
-
lines?: Array<string>;
|
|
25
|
-
}>;
|
|
26
|
-
}
|
|
27
|
-
/**
|
|
28
|
-
* Provider options common to all ElevenLabs audio endpoints.
|
|
29
|
-
*/
|
|
30
|
-
interface CommonAudioOptions {
|
|
31
|
-
/** Output audio format. Defaults to `mp3_44100_128`. */
|
|
32
|
-
outputFormat?: ElevenLabsOutputFormat;
|
|
33
|
-
}
|
|
34
|
-
/**
|
|
35
|
-
* Provider options for music generation (`music_v1`).
|
|
36
|
-
*/
|
|
37
|
-
export interface ElevenLabsMusicProviderOptions extends CommonAudioOptions {
|
|
38
|
-
/** Structured composition plan. Mutually exclusive with `prompt`/`duration`. */
|
|
39
|
-
compositionPlan?: ElevenLabsMusicCompositionPlan;
|
|
40
|
-
/** Deterministic sampling seed (incompatible with `prompt`). */
|
|
41
|
-
seed?: number;
|
|
42
|
-
/** Force the output to be purely instrumental (prompt-mode only). */
|
|
43
|
-
forceInstrumental?: boolean;
|
|
44
|
-
/** Strictly respect section durations in `compositionPlan`. */
|
|
45
|
-
respectSectionsDurations?: boolean;
|
|
46
|
-
}
|
|
47
|
-
/**
|
|
48
|
-
* Provider options for sound-effect generation (`eleven_text_to_sound_v*`).
|
|
49
|
-
*/
|
|
50
|
-
export interface ElevenLabsSoundEffectsProviderOptions extends CommonAudioOptions {
|
|
51
|
-
/** Prompt influence, 0..1. Default 0.3. Higher = more prompt adherence. */
|
|
52
|
-
promptInfluence?: number;
|
|
53
|
-
/** Generate a loopable SFX (v2 only). */
|
|
54
|
-
loop?: boolean;
|
|
55
|
-
}
|
|
56
|
-
/**
|
|
57
|
-
* Union of per-model provider options. We keep both branches on one type so
|
|
58
|
-
* the adapter stays tree-shakeable; callers narrow by model at the factory.
|
|
59
|
-
*/
|
|
60
|
-
export type ElevenLabsAudioProviderOptions = (ElevenLabsMusicProviderOptions & ElevenLabsSoundEffectsProviderOptions) | ElevenLabsMusicProviderOptions | ElevenLabsSoundEffectsProviderOptions;
|
|
61
|
-
/**
|
|
62
|
-
* ElevenLabs audio generation adapter. Dispatches to music or SFX endpoints
|
|
63
|
-
* based on the model id. Music → `client.music.compose`, SFX →
|
|
64
|
-
* `client.textToSoundEffects.convert`.
|
|
65
|
-
*
|
|
66
|
-
* @example
|
|
67
|
-
* ```ts
|
|
68
|
-
* const music = elevenlabsAudio('music_v1')
|
|
69
|
-
* await generateAudio({ adapter: music, prompt: 'lo-fi beat', duration: 15 })
|
|
70
|
-
*
|
|
71
|
-
* const sfx = elevenlabsAudio('eleven_text_to_sound_v2')
|
|
72
|
-
* await generateAudio({ adapter: sfx, prompt: 'glass shattering', duration: 3 })
|
|
73
|
-
* ```
|
|
74
|
-
*/
|
|
75
|
-
export declare class ElevenLabsAudioAdapter<TModel extends ElevenLabsAudioModel> extends BaseAudioAdapter<TModel, ElevenLabsAudioProviderOptions> {
|
|
76
|
-
readonly name: "elevenlabs";
|
|
77
|
-
private readonly client;
|
|
78
|
-
constructor(model: TModel, config?: ElevenLabsClientConfig);
|
|
79
|
-
generateAudio(options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>): Promise<AudioGenerationResult>;
|
|
80
|
-
private runMusic;
|
|
81
|
-
private runSoundEffects;
|
|
82
|
-
private finalize;
|
|
83
|
-
protected generateId(): string;
|
|
84
|
-
}
|
|
85
|
-
/**
|
|
86
|
-
* Create an ElevenLabs audio adapter using `ELEVENLABS_API_KEY` from env.
|
|
87
|
-
*/
|
|
88
|
-
export declare function elevenlabsAudio<TModel extends ElevenLabsAudioModel>(model: TModel, config?: ElevenLabsClientConfig): ElevenLabsAudioAdapter<TModel>;
|
|
89
|
-
/**
|
|
90
|
-
* Create an ElevenLabs audio adapter with an explicit API key.
|
|
91
|
-
*/
|
|
92
|
-
export declare function createElevenLabsAudio<TModel extends ElevenLabsAudioModel>(model: TModel, apiKey: string, config?: Omit<ElevenLabsClientConfig, 'apiKey'>): ElevenLabsAudioAdapter<TModel>;
|
|
93
|
-
export {};
|
|
@@ -1,107 +0,0 @@
|
|
|
1
|
-
import { BaseAudioAdapter } from "@tanstack/ai/adapters";
|
|
2
|
-
import { createElevenLabsClient, readStreamToArrayBuffer, arrayBufferToBase64, parseOutputFormat, generateId } from "../utils/client.js";
|
|
3
|
-
import { isElevenLabsMusicModel, isElevenLabsSoundEffectsModel } from "../model-meta.js";
|
|
4
|
-
class ElevenLabsAudioAdapter extends BaseAudioAdapter {
|
|
5
|
-
name = "elevenlabs";
|
|
6
|
-
client;
|
|
7
|
-
constructor(model, config) {
|
|
8
|
-
super(model, config ?? {});
|
|
9
|
-
this.client = createElevenLabsClient(config);
|
|
10
|
-
}
|
|
11
|
-
async generateAudio(options) {
|
|
12
|
-
const { logger } = options;
|
|
13
|
-
logger.request(
|
|
14
|
-
`activity=generateAudio provider=elevenlabs model=${this.model}`,
|
|
15
|
-
{ provider: "elevenlabs", model: this.model }
|
|
16
|
-
);
|
|
17
|
-
try {
|
|
18
|
-
if (isElevenLabsMusicModel(this.model)) {
|
|
19
|
-
return await this.runMusic(options);
|
|
20
|
-
}
|
|
21
|
-
if (isElevenLabsSoundEffectsModel(this.model)) {
|
|
22
|
-
return await this.runSoundEffects(options);
|
|
23
|
-
}
|
|
24
|
-
throw new Error(
|
|
25
|
-
`Unsupported ElevenLabs audio model "${this.model}". Expected one of: music_v1, eleven_text_to_sound_v2, eleven_text_to_sound_v1.`
|
|
26
|
-
);
|
|
27
|
-
} catch (error) {
|
|
28
|
-
logger.errors("elevenlabs.generateAudio fatal", {
|
|
29
|
-
error,
|
|
30
|
-
source: "elevenlabs.generateAudio"
|
|
31
|
-
});
|
|
32
|
-
throw error;
|
|
33
|
-
}
|
|
34
|
-
}
|
|
35
|
-
async runMusic(options) {
|
|
36
|
-
const modelId = this.model;
|
|
37
|
-
const music = options.modelOptions ?? {};
|
|
38
|
-
const outputFormat = music.outputFormat;
|
|
39
|
-
const stream = await this.client.music.compose({
|
|
40
|
-
modelId,
|
|
41
|
-
...options.prompt && !music.compositionPlan ? { prompt: options.prompt } : {},
|
|
42
|
-
...music.compositionPlan ? { compositionPlan: toMusicPrompt(music.compositionPlan) } : {},
|
|
43
|
-
...options.duration != null && !music.compositionPlan ? { musicLengthMs: Math.round(options.duration * 1e3) } : {},
|
|
44
|
-
...outputFormat ? { outputFormat } : {},
|
|
45
|
-
...music.seed != null ? { seed: music.seed } : {},
|
|
46
|
-
...music.forceInstrumental != null ? { forceInstrumental: music.forceInstrumental } : {},
|
|
47
|
-
...music.respectSectionsDurations != null ? { respectSectionsDurations: music.respectSectionsDurations } : {}
|
|
48
|
-
});
|
|
49
|
-
return this.finalize(stream, outputFormat, options.duration);
|
|
50
|
-
}
|
|
51
|
-
async runSoundEffects(options) {
|
|
52
|
-
const modelId = this.model;
|
|
53
|
-
const sfx = options.modelOptions ?? {};
|
|
54
|
-
const outputFormat = sfx.outputFormat;
|
|
55
|
-
const stream = await this.client.textToSoundEffects.convert({
|
|
56
|
-
text: options.prompt,
|
|
57
|
-
modelId,
|
|
58
|
-
...options.duration != null ? { durationSeconds: options.duration } : {},
|
|
59
|
-
...outputFormat ? { outputFormat } : {},
|
|
60
|
-
...sfx.promptInfluence != null ? { promptInfluence: sfx.promptInfluence } : {},
|
|
61
|
-
...sfx.loop != null ? { loop: sfx.loop } : {}
|
|
62
|
-
});
|
|
63
|
-
return this.finalize(stream, outputFormat, options.duration);
|
|
64
|
-
}
|
|
65
|
-
async finalize(stream, outputFormat, duration) {
|
|
66
|
-
const buffer = await readStreamToArrayBuffer(stream);
|
|
67
|
-
const base64 = arrayBufferToBase64(buffer);
|
|
68
|
-
const { contentType } = parseOutputFormat(outputFormat);
|
|
69
|
-
return {
|
|
70
|
-
id: generateId(this.name),
|
|
71
|
-
model: this.model,
|
|
72
|
-
audio: {
|
|
73
|
-
b64Json: base64,
|
|
74
|
-
contentType,
|
|
75
|
-
...duration != null ? { duration } : {}
|
|
76
|
-
}
|
|
77
|
-
};
|
|
78
|
-
}
|
|
79
|
-
generateId() {
|
|
80
|
-
return generateId(this.name);
|
|
81
|
-
}
|
|
82
|
-
}
|
|
83
|
-
function toMusicPrompt(plan) {
|
|
84
|
-
return {
|
|
85
|
-
positiveGlobalStyles: plan.positiveGlobalStyles ?? [],
|
|
86
|
-
negativeGlobalStyles: plan.negativeGlobalStyles ?? [],
|
|
87
|
-
sections: (plan.sections ?? []).map((section) => ({
|
|
88
|
-
sectionName: section.sectionName,
|
|
89
|
-
positiveLocalStyles: section.positiveLocalStyles ?? [],
|
|
90
|
-
negativeLocalStyles: section.negativeLocalStyles ?? [],
|
|
91
|
-
durationMs: section.durationMs ?? 1e4,
|
|
92
|
-
lines: section.lines ?? []
|
|
93
|
-
}))
|
|
94
|
-
};
|
|
95
|
-
}
|
|
96
|
-
function elevenlabsAudio(model, config) {
|
|
97
|
-
return new ElevenLabsAudioAdapter(model, config);
|
|
98
|
-
}
|
|
99
|
-
function createElevenLabsAudio(model, apiKey, config) {
|
|
100
|
-
return new ElevenLabsAudioAdapter(model, { apiKey, ...config });
|
|
101
|
-
}
|
|
102
|
-
export {
|
|
103
|
-
ElevenLabsAudioAdapter,
|
|
104
|
-
createElevenLabsAudio,
|
|
105
|
-
elevenlabsAudio
|
|
106
|
-
};
|
|
107
|
-
//# sourceMappingURL=audio.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"audio.js","sources":["../../../src/adapters/audio.ts"],"sourcesContent":["import { BaseAudioAdapter } from '@tanstack/ai/adapters'\nimport {\n arrayBufferToBase64,\n createElevenLabsClient,\n generateId,\n parseOutputFormat,\n readStreamToArrayBuffer,\n} from '../utils/client'\nimport {\n isElevenLabsMusicModel,\n isElevenLabsSoundEffectsModel,\n} from '../model-meta'\nimport type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'\nimport type {\n AudioGenerationOptions,\n AudioGenerationResult,\n} from '@tanstack/ai'\nimport type { ElevenLabsClientConfig } from '../utils/client'\nimport type {\n ElevenLabsAudioModel,\n ElevenLabsMusicModel,\n ElevenLabsOutputFormat,\n ElevenLabsSoundEffectsModel,\n} from '../model-meta'\n\n/**\n * Structured composition plan for ElevenLabs music generation. Mutually\n * exclusive with a free-form `prompt` on the `generateAudio()` call — when\n * supplied, `prompt` is ignored by ElevenLabs.\n *\n * We mirror the SDK's camelCase naming. Lengths are in milliseconds.\n * @see https://elevenlabs.io/docs/api-reference/music/compose\n */\nexport interface ElevenLabsMusicCompositionPlan {\n /** Positive global style descriptors (mood, instruments, tempo, …). */\n positiveGlobalStyles?: Array<string>\n /** Negative global style descriptors — styles to avoid. */\n negativeGlobalStyles?: Array<string>\n /** Section definitions (verse/chorus/bridge/…) with local style hints. */\n sections?: Array<{\n sectionName: string\n positiveLocalStyles?: Array<string>\n negativeLocalStyles?: Array<string>\n durationMs?: number\n lines?: Array<string>\n }>\n}\n\n/**\n * Provider options common to all ElevenLabs audio endpoints.\n */\ninterface CommonAudioOptions {\n /** Output audio format. Defaults to `mp3_44100_128`. */\n outputFormat?: ElevenLabsOutputFormat\n}\n\n/**\n * Provider options for music generation (`music_v1`).\n */\nexport interface ElevenLabsMusicProviderOptions extends CommonAudioOptions {\n /** Structured composition plan. Mutually exclusive with `prompt`/`duration`. */\n compositionPlan?: ElevenLabsMusicCompositionPlan\n /** Deterministic sampling seed (incompatible with `prompt`). */\n seed?: number\n /** Force the output to be purely instrumental (prompt-mode only). */\n forceInstrumental?: boolean\n /** Strictly respect section durations in `compositionPlan`. */\n respectSectionsDurations?: boolean\n}\n\n/**\n * Provider options for sound-effect generation (`eleven_text_to_sound_v*`).\n */\nexport interface ElevenLabsSoundEffectsProviderOptions extends CommonAudioOptions {\n /** Prompt influence, 0..1. Default 0.3. Higher = more prompt adherence. */\n promptInfluence?: number\n /** Generate a loopable SFX (v2 only). */\n loop?: boolean\n}\n\n/**\n * Union of per-model provider options. We keep both branches on one type so\n * the adapter stays tree-shakeable; callers narrow by model at the factory.\n */\nexport type ElevenLabsAudioProviderOptions =\n | (ElevenLabsMusicProviderOptions & ElevenLabsSoundEffectsProviderOptions)\n | ElevenLabsMusicProviderOptions\n | ElevenLabsSoundEffectsProviderOptions\n\n/**\n * ElevenLabs audio generation adapter. Dispatches to music or SFX endpoints\n * based on the model id. Music → `client.music.compose`, SFX →\n * `client.textToSoundEffects.convert`.\n *\n * @example\n * ```ts\n * const music = elevenlabsAudio('music_v1')\n * await generateAudio({ adapter: music, prompt: 'lo-fi beat', duration: 15 })\n *\n * const sfx = elevenlabsAudio('eleven_text_to_sound_v2')\n * await generateAudio({ adapter: sfx, prompt: 'glass shattering', duration: 3 })\n * ```\n */\nexport class ElevenLabsAudioAdapter<\n TModel extends ElevenLabsAudioModel,\n> extends BaseAudioAdapter<TModel, ElevenLabsAudioProviderOptions> {\n readonly name = 'elevenlabs' as const\n\n private readonly client: ElevenLabsClient\n\n constructor(model: TModel, config?: ElevenLabsClientConfig) {\n super(model, config ?? {})\n this.client = createElevenLabsClient(config)\n }\n\n async generateAudio(\n options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,\n ): Promise<AudioGenerationResult> {\n const { logger } = options\n logger.request(\n `activity=generateAudio provider=elevenlabs model=${this.model}`,\n { provider: 'elevenlabs', model: this.model },\n )\n try {\n if (isElevenLabsMusicModel(this.model)) {\n return await this.runMusic(options)\n }\n if (isElevenLabsSoundEffectsModel(this.model)) {\n return await this.runSoundEffects(options)\n }\n throw new Error(\n `Unsupported ElevenLabs audio model \"${this.model}\". Expected one of: music_v1, eleven_text_to_sound_v2, eleven_text_to_sound_v1.`,\n )\n } catch (error) {\n logger.errors('elevenlabs.generateAudio fatal', {\n error,\n source: 'elevenlabs.generateAudio',\n })\n throw error\n }\n }\n\n private async runMusic(\n options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,\n ): Promise<AudioGenerationResult> {\n // Gated by isElevenLabsMusicModel() in generateAudio().\n const modelId = this.model as ElevenLabsMusicModel\n const music = (options.modelOptions ?? {}) as ElevenLabsMusicProviderOptions\n const outputFormat = music.outputFormat\n\n const stream = await this.client.music.compose({\n modelId,\n ...(options.prompt && !music.compositionPlan\n ? { prompt: options.prompt }\n : {}),\n ...(music.compositionPlan\n ? { compositionPlan: toMusicPrompt(music.compositionPlan) }\n : {}),\n ...(options.duration != null && !music.compositionPlan\n ? { musicLengthMs: Math.round(options.duration * 1000) }\n : {}),\n ...(outputFormat ? { outputFormat } : {}),\n ...(music.seed != null ? { seed: music.seed } : {}),\n ...(music.forceInstrumental != null\n ? { forceInstrumental: music.forceInstrumental }\n : {}),\n ...(music.respectSectionsDurations != null\n ? { respectSectionsDurations: music.respectSectionsDurations }\n : {}),\n })\n\n return this.finalize(stream, outputFormat, options.duration)\n }\n\n private async runSoundEffects(\n options: AudioGenerationOptions<ElevenLabsAudioProviderOptions>,\n ): Promise<AudioGenerationResult> {\n // Gated by isElevenLabsSoundEffectsModel() in generateAudio().\n const modelId = this.model as ElevenLabsSoundEffectsModel\n const sfx = (options.modelOptions ??\n {}) as ElevenLabsSoundEffectsProviderOptions\n const outputFormat = sfx.outputFormat\n\n const stream = await this.client.textToSoundEffects.convert({\n text: options.prompt,\n modelId,\n ...(options.duration != null\n ? { durationSeconds: options.duration }\n : {}),\n ...(outputFormat ? { outputFormat } : {}),\n ...(sfx.promptInfluence != null\n ? { promptInfluence: sfx.promptInfluence }\n : {}),\n ...(sfx.loop != null ? { loop: sfx.loop } : {}),\n })\n\n return this.finalize(stream, outputFormat, options.duration)\n }\n\n private async finalize(\n stream: ReadableStream<Uint8Array>,\n outputFormat: ElevenLabsOutputFormat | undefined,\n duration: number | undefined,\n ): Promise<AudioGenerationResult> {\n const buffer = await readStreamToArrayBuffer(stream)\n const base64 = arrayBufferToBase64(buffer)\n const { contentType } = parseOutputFormat(outputFormat)\n return {\n id: generateId(this.name),\n model: this.model,\n audio: {\n b64Json: base64,\n contentType,\n ...(duration != null ? { duration } : {}),\n },\n }\n }\n\n protected override generateId(): string {\n return generateId(this.name)\n }\n}\n\nfunction toMusicPrompt(plan: ElevenLabsMusicCompositionPlan) {\n return {\n positiveGlobalStyles: plan.positiveGlobalStyles ?? [],\n negativeGlobalStyles: plan.negativeGlobalStyles ?? [],\n sections: (plan.sections ?? []).map((section) => ({\n sectionName: section.sectionName,\n positiveLocalStyles: section.positiveLocalStyles ?? [],\n negativeLocalStyles: section.negativeLocalStyles ?? [],\n durationMs: section.durationMs ?? 10000,\n lines: section.lines ?? [],\n })),\n }\n}\n\n/**\n * Create an ElevenLabs audio adapter using `ELEVENLABS_API_KEY` from env.\n */\nexport function elevenlabsAudio<TModel extends ElevenLabsAudioModel>(\n model: TModel,\n config?: ElevenLabsClientConfig,\n): ElevenLabsAudioAdapter<TModel> {\n return new ElevenLabsAudioAdapter(model, config)\n}\n\n/**\n * Create an ElevenLabs audio adapter with an explicit API key.\n */\nexport function createElevenLabsAudio<TModel extends ElevenLabsAudioModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<ElevenLabsClientConfig, 'apiKey'>,\n): ElevenLabsAudioAdapter<TModel> {\n return new ElevenLabsAudioAdapter(model, { apiKey, ...config })\n}\n"],"names":[],"mappings":";;;AAuGO,MAAM,+BAEH,iBAAyD;AAAA,EACxD,OAAO;AAAA,EAEC;AAAA,EAEjB,YAAY,OAAe,QAAiC;AAC1D,UAAM,OAAO,UAAU,EAAE;AACzB,SAAK,SAAS,uBAAuB,MAAM;AAAA,EAC7C;AAAA,EAEA,MAAM,cACJ,SACgC;AAChC,UAAM,EAAE,WAAW;AACnB,WAAO;AAAA,MACL,oDAAoD,KAAK,KAAK;AAAA,MAC9D,EAAE,UAAU,cAAc,OAAO,KAAK,MAAA;AAAA,IAAM;AAE9C,QAAI;AACF,UAAI,uBAAuB,KAAK,KAAK,GAAG;AACtC,eAAO,MAAM,KAAK,SAAS,OAAO;AAAA,MACpC;AACA,UAAI,8BAA8B,KAAK,KAAK,GAAG;AAC7C,eAAO,MAAM,KAAK,gBAAgB,OAAO;AAAA,MAC3C;AACA,YAAM,IAAI;AAAA,QACR,uCAAuC,KAAK,KAAK;AAAA,MAAA;AAAA,IAErD,SAAS,OAAO;AACd,aAAO,OAAO,kCAAkC;AAAA,QAC9C;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AAAA,EAEA,MAAc,SACZ,SACgC;AAEhC,UAAM,UAAU,KAAK;AACrB,UAAM,QAAS,QAAQ,gBAAgB,CAAA;AACvC,UAAM,eAAe,MAAM;AAE3B,UAAM,SAAS,MAAM,KAAK,OAAO,MAAM,QAAQ;AAAA,MAC7C;AAAA,MACA,GAAI,QAAQ,UAAU,CAAC,MAAM,kBACzB,EAAE,QAAQ,QAAQ,OAAA,IAClB,CAAA;AAAA,MACJ,GAAI,MAAM,kBACN,EAAE,iBAAiB,cAAc,MAAM,eAAe,EAAA,IACtD,CAAA;AAAA,MACJ,GAAI,QAAQ,YAAY,QAAQ,CAAC,MAAM,kBACnC,EAAE,eAAe,KAAK,MAAM,QAAQ,WAAW,GAAI,EAAA,IACnD,CAAA;AAAA,MACJ,GAAI,eAAe,EAAE,aAAA,IAAiB,CAAA;AAAA,MACtC,GAAI,MAAM,QAAQ,OAAO,EAAE,MAAM,MAAM,KAAA,IAAS,CAAA;AAAA,MAChD,GAAI,MAAM,qBAAqB,OAC3B,EAAE,mBAAmB,MAAM,kBAAA,IAC3B,CAAA;AAAA,MACJ,GAAI,MAAM,4BAA4B,OAClC,EAAE,0BAA0B,MAAM,6BAClC,CAAA;AAAA,IAAC,CACN;AAED,WAAO,KAAK,SAAS,QAAQ,cAAc,QAAQ,QAAQ;AAAA,EAC7D;AAAA,EAEA,MAAc,gBACZ,SACgC;AAEhC,UAAM,UAAU,KAAK;AACrB,UAAM,MAAO,QAAQ,gBACnB,CAAA;AACF,UAAM,eAAe,IAAI;AAEzB,UAAM,SAAS,MAAM,KAAK,OAAO,mBAAmB,QAAQ;AAAA,MAC1D,MAAM,QAAQ;AAAA,MACd;AAAA,MACA,GAAI,QAAQ,YAAY,OACpB,EAAE,iBAAiB,QAAQ,SAAA,IAC3B,CAAA;AAAA,MACJ,GAAI,eAAe,EAAE,aAAA,IAAiB,CAAA;AAAA,MACtC,GAAI,IAAI,mBAAmB,OACvB,EAAE,iBAAiB,IAAI,gBAAA,IACvB,CAAA;AAAA,MACJ,GAAI,IAAI,QAAQ,OAAO,EAAE,MAAM,IAAI,SAAS,CAAA;AAAA,IAAC,CAC9C;AAED,WAAO,KAAK,SAAS,QAAQ,cAAc,QAAQ,QAAQ;AAAA,EAC7D;AAAA,EAEA,MAAc,SACZ,QACA,cACA,UACgC;AAChC,UAAM,SAAS,MAAM,wBAAwB,MAAM;AACnD,UAAM,SAAS,oBAAoB,MAAM;AACzC,UAAM,EAAE,YAAA,IAAgB,kBAAkB,YAAY;AACtD,WAAO;AAAA,MACL,IAAI,WAAW,KAAK,IAAI;AAAA,MACxB,OAAO,KAAK;AAAA,MACZ,OAAO;AAAA,QACL,SAAS;AAAA,QACT;AAAA,QACA,GAAI,YAAY,OAAO,EAAE,aAAa,CAAA;AAAA,MAAC;AAAA,IACzC;AAAA,EAEJ;AAAA,EAEmB,aAAqB;AACtC,WAAO,WAAW,KAAK,IAAI;AAAA,EAC7B;AACF;AAEA,SAAS,cAAc,MAAsC;AAC3D,SAAO;AAAA,IACL,sBAAsB,KAAK,wBAAwB,CAAA;AAAA,IACnD,sBAAsB,KAAK,wBAAwB,CAAA;AAAA,IACnD,WAAW,KAAK,YAAY,CAAA,GAAI,IAAI,CAAC,aAAa;AAAA,MAChD,aAAa,QAAQ;AAAA,MACrB,qBAAqB,QAAQ,uBAAuB,CAAA;AAAA,MACpD,qBAAqB,QAAQ,uBAAuB,CAAA;AAAA,MACpD,YAAY,QAAQ,cAAc;AAAA,MAClC,OAAO,QAAQ,SAAS,CAAA;AAAA,IAAC,EACzB;AAAA,EAAA;AAEN;AAKO,SAAS,gBACd,OACA,QACgC;AAChC,SAAO,IAAI,uBAAuB,OAAO,MAAM;AACjD;AAKO,SAAS,sBACd,OACA,QACA,QACgC;AAChC,SAAO,IAAI,uBAAuB,OAAO,EAAE,QAAQ,GAAG,QAAQ;AAChE;"}
|
|
@@ -1,83 +0,0 @@
|
|
|
1
|
-
import { BaseTTSAdapter } from '@tanstack/ai/adapters';
|
|
2
|
-
import { TTSOptions, TTSResult } from '@tanstack/ai';
|
|
3
|
-
import { ElevenLabsClientConfig } from '../utils/client.js';
|
|
4
|
-
import { ElevenLabsOutputFormat, ElevenLabsTTSModel } from '../model-meta.js';
|
|
5
|
-
/**
|
|
6
|
-
* ElevenLabs voice settings overrides. All fields are optional — omitted
|
|
7
|
-
* values fall back to the voice's stored defaults.
|
|
8
|
-
* @see https://elevenlabs.io/docs/api-reference/text-to-speech/convert
|
|
9
|
-
*/
|
|
10
|
-
export interface ElevenLabsVoiceSettings {
|
|
11
|
-
/** Voice stability, 0..1. Default 0.5. */
|
|
12
|
-
stability?: number;
|
|
13
|
-
/** Similarity boost, 0..1. Default 0.75. */
|
|
14
|
-
similarityBoost?: number;
|
|
15
|
-
/** Style exaggeration, 0..1. Default 0. */
|
|
16
|
-
style?: number;
|
|
17
|
-
/** Playback speed. Default 1.0. */
|
|
18
|
-
speed?: number;
|
|
19
|
-
/** Clarity/presence boost. Default true. */
|
|
20
|
-
useSpeakerBoost?: boolean;
|
|
21
|
-
}
|
|
22
|
-
/**
|
|
23
|
-
* Provider-specific TTS options. `voice` on `generateSpeech()` takes priority
|
|
24
|
-
* over `voiceId` here, but we expose the same field for callers that prefer
|
|
25
|
-
* to keep voice configuration inside the adapter config.
|
|
26
|
-
*/
|
|
27
|
-
export interface ElevenLabsSpeechProviderOptions {
|
|
28
|
-
/** ElevenLabs voice ID to synthesize. Required if `generateSpeech().voice` is not set. */
|
|
29
|
-
voiceId?: string;
|
|
30
|
-
/** Output audio format encoded as `codec_samplerate[_bitrate]`. Defaults to `mp3_44100_128`. */
|
|
31
|
-
outputFormat?: ElevenLabsOutputFormat;
|
|
32
|
-
/** Voice-settings overrides for this request only. */
|
|
33
|
-
voiceSettings?: ElevenLabsVoiceSettings;
|
|
34
|
-
/** ISO-639-1 language code to enforce (e.g. `'en'`, `'ja'`). */
|
|
35
|
-
languageCode?: string;
|
|
36
|
-
/** Deterministic sampling seed, 0..4294967295. */
|
|
37
|
-
seed?: number;
|
|
38
|
-
/** Previous text for stitching adjacent clips. */
|
|
39
|
-
previousText?: string;
|
|
40
|
-
/** Next text for stitching adjacent clips. */
|
|
41
|
-
nextText?: string;
|
|
42
|
-
/** Previous request IDs for stitching (max 3). */
|
|
43
|
-
previousRequestIds?: Array<string>;
|
|
44
|
-
/** Next request IDs for stitching (max 3). */
|
|
45
|
-
nextRequestIds?: Array<string>;
|
|
46
|
-
/** Text normalization toggle. Default `'auto'`. */
|
|
47
|
-
applyTextNormalization?: 'auto' | 'on' | 'off';
|
|
48
|
-
/** Language-specific text normalization (currently Japanese only, adds latency). */
|
|
49
|
-
applyLanguageTextNormalization?: boolean;
|
|
50
|
-
/** Latency optimization level, 0..4. */
|
|
51
|
-
optimizeStreamingLatency?: number;
|
|
52
|
-
/** Enable logging. Set false for zero-retention mode (enterprise only). */
|
|
53
|
-
enableLogging?: boolean;
|
|
54
|
-
}
|
|
55
|
-
/**
|
|
56
|
-
* ElevenLabs text-to-speech adapter built on the official
|
|
57
|
-
* `@elevenlabs/elevenlabs-js` SDK.
|
|
58
|
-
*
|
|
59
|
-
* @example
|
|
60
|
-
* ```ts
|
|
61
|
-
* const adapter = elevenlabsSpeech('eleven_multilingual_v2')
|
|
62
|
-
* const result = await generateSpeech({
|
|
63
|
-
* adapter,
|
|
64
|
-
* text: 'Hello, world!',
|
|
65
|
-
* voice: '21m00Tcm4TlvDq8ikWAM',
|
|
66
|
-
* })
|
|
67
|
-
* ```
|
|
68
|
-
*/
|
|
69
|
-
export declare class ElevenLabsSpeechAdapter<TModel extends ElevenLabsTTSModel> extends BaseTTSAdapter<TModel, ElevenLabsSpeechProviderOptions> {
|
|
70
|
-
readonly name: "elevenlabs";
|
|
71
|
-
private readonly client;
|
|
72
|
-
constructor(model: TModel, config?: ElevenLabsClientConfig);
|
|
73
|
-
generateSpeech(options: TTSOptions<ElevenLabsSpeechProviderOptions>): Promise<TTSResult>;
|
|
74
|
-
protected generateId(): string;
|
|
75
|
-
}
|
|
76
|
-
/**
|
|
77
|
-
* Create an ElevenLabs speech adapter using `ELEVENLABS_API_KEY` from env.
|
|
78
|
-
*/
|
|
79
|
-
export declare function elevenlabsSpeech<TModel extends ElevenLabsTTSModel>(model: TModel, config?: ElevenLabsClientConfig): ElevenLabsSpeechAdapter<TModel>;
|
|
80
|
-
/**
|
|
81
|
-
* Create an ElevenLabs speech adapter with an explicit API key.
|
|
82
|
-
*/
|
|
83
|
-
export declare function createElevenLabsSpeech<TModel extends ElevenLabsTTSModel>(model: TModel, apiKey: string, config?: Omit<ElevenLabsClientConfig, 'apiKey'>): ElevenLabsSpeechAdapter<TModel>;
|
|
@@ -1,113 +0,0 @@
|
|
|
1
|
-
import { BaseTTSAdapter } from "@tanstack/ai/adapters";
|
|
2
|
-
import { createElevenLabsClient, readStreamToArrayBuffer, arrayBufferToBase64, parseOutputFormat, generateId } from "../utils/client.js";
|
|
3
|
-
class ElevenLabsSpeechAdapter extends BaseTTSAdapter {
|
|
4
|
-
name = "elevenlabs";
|
|
5
|
-
client;
|
|
6
|
-
constructor(model, config) {
|
|
7
|
-
super(model, config ?? {});
|
|
8
|
-
this.client = createElevenLabsClient(config);
|
|
9
|
-
}
|
|
10
|
-
async generateSpeech(options) {
|
|
11
|
-
const { logger } = options;
|
|
12
|
-
logger.request(
|
|
13
|
-
`activity=generateSpeech provider=elevenlabs model=${this.model}`,
|
|
14
|
-
{ provider: "elevenlabs", model: this.model }
|
|
15
|
-
);
|
|
16
|
-
try {
|
|
17
|
-
const voiceId = options.voice ?? options.modelOptions?.voiceId;
|
|
18
|
-
if (!voiceId) {
|
|
19
|
-
throw new Error(
|
|
20
|
-
"ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions."
|
|
21
|
-
);
|
|
22
|
-
}
|
|
23
|
-
const {
|
|
24
|
-
outputFormat,
|
|
25
|
-
voiceSettings,
|
|
26
|
-
languageCode,
|
|
27
|
-
seed,
|
|
28
|
-
previousText,
|
|
29
|
-
nextText,
|
|
30
|
-
previousRequestIds,
|
|
31
|
-
nextRequestIds,
|
|
32
|
-
applyTextNormalization,
|
|
33
|
-
applyLanguageTextNormalization,
|
|
34
|
-
optimizeStreamingLatency,
|
|
35
|
-
enableLogging
|
|
36
|
-
} = options.modelOptions ?? {};
|
|
37
|
-
const effectiveOutputFormat = outputFormat ?? inferOutputFormatFromResponseFormat(options.format);
|
|
38
|
-
const stream = await this.client.textToSpeech.convert(voiceId, {
|
|
39
|
-
text: options.text,
|
|
40
|
-
modelId: this.model,
|
|
41
|
-
...effectiveOutputFormat ? { outputFormat: effectiveOutputFormat } : {},
|
|
42
|
-
...voiceSettings ? { voiceSettings: mapVoiceSettings(voiceSettings, options.speed) } : options.speed != null ? { voiceSettings: { speed: options.speed } } : {},
|
|
43
|
-
...languageCode ? { languageCode } : {},
|
|
44
|
-
...seed != null ? { seed } : {},
|
|
45
|
-
...previousText ? { previousText } : {},
|
|
46
|
-
...nextText ? { nextText } : {},
|
|
47
|
-
...previousRequestIds ? { previousRequestIds } : {},
|
|
48
|
-
...nextRequestIds ? { nextRequestIds } : {},
|
|
49
|
-
...applyTextNormalization ? { applyTextNormalization } : {},
|
|
50
|
-
...applyLanguageTextNormalization != null ? { applyLanguageTextNormalization } : {},
|
|
51
|
-
...optimizeStreamingLatency != null ? { optimizeStreamingLatency } : {},
|
|
52
|
-
...enableLogging != null ? { enableLogging } : {}
|
|
53
|
-
});
|
|
54
|
-
const buffer = await readStreamToArrayBuffer(stream);
|
|
55
|
-
const base64 = arrayBufferToBase64(buffer);
|
|
56
|
-
const { format, contentType } = parseOutputFormat(effectiveOutputFormat);
|
|
57
|
-
return {
|
|
58
|
-
id: generateId(this.name),
|
|
59
|
-
model: this.model,
|
|
60
|
-
audio: base64,
|
|
61
|
-
format,
|
|
62
|
-
contentType
|
|
63
|
-
};
|
|
64
|
-
} catch (error) {
|
|
65
|
-
logger.errors("elevenlabs.generateSpeech fatal", {
|
|
66
|
-
error,
|
|
67
|
-
source: "elevenlabs.generateSpeech"
|
|
68
|
-
});
|
|
69
|
-
throw error;
|
|
70
|
-
}
|
|
71
|
-
}
|
|
72
|
-
generateId() {
|
|
73
|
-
return generateId(this.name);
|
|
74
|
-
}
|
|
75
|
-
}
|
|
76
|
-
function mapVoiceSettings(settings, speedOverride) {
|
|
77
|
-
return {
|
|
78
|
-
...settings.stability != null ? { stability: settings.stability } : {},
|
|
79
|
-
...settings.similarityBoost != null ? { similarityBoost: settings.similarityBoost } : {},
|
|
80
|
-
...settings.style != null ? { style: settings.style } : {},
|
|
81
|
-
...speedOverride != null ? { speed: speedOverride } : settings.speed != null ? { speed: settings.speed } : {},
|
|
82
|
-
...settings.useSpeakerBoost != null ? { useSpeakerBoost: settings.useSpeakerBoost } : {}
|
|
83
|
-
};
|
|
84
|
-
}
|
|
85
|
-
function inferOutputFormatFromResponseFormat(format) {
|
|
86
|
-
switch (format) {
|
|
87
|
-
case "mp3":
|
|
88
|
-
return "mp3_44100_128";
|
|
89
|
-
case "pcm":
|
|
90
|
-
return "pcm_44100";
|
|
91
|
-
case "opus":
|
|
92
|
-
return "opus_48000_128";
|
|
93
|
-
case void 0:
|
|
94
|
-
return void 0;
|
|
95
|
-
case "aac":
|
|
96
|
-
case "flac":
|
|
97
|
-
case "wav":
|
|
98
|
-
default:
|
|
99
|
-
return "mp3_44100_128";
|
|
100
|
-
}
|
|
101
|
-
}
|
|
102
|
-
function elevenlabsSpeech(model, config) {
|
|
103
|
-
return new ElevenLabsSpeechAdapter(model, config);
|
|
104
|
-
}
|
|
105
|
-
function createElevenLabsSpeech(model, apiKey, config) {
|
|
106
|
-
return new ElevenLabsSpeechAdapter(model, { apiKey, ...config });
|
|
107
|
-
}
|
|
108
|
-
export {
|
|
109
|
-
ElevenLabsSpeechAdapter,
|
|
110
|
-
createElevenLabsSpeech,
|
|
111
|
-
elevenlabsSpeech
|
|
112
|
-
};
|
|
113
|
-
//# sourceMappingURL=speech.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"speech.js","sources":["../../../src/adapters/speech.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport {\n arrayBufferToBase64,\n createElevenLabsClient,\n generateId,\n parseOutputFormat,\n readStreamToArrayBuffer,\n} from '../utils/client'\nimport type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'\nimport type { TTSOptions, TTSResult } from '@tanstack/ai'\nimport type { ElevenLabsClientConfig } from '../utils/client'\nimport type { ElevenLabsOutputFormat, ElevenLabsTTSModel } from '../model-meta'\n\n/**\n * ElevenLabs voice settings overrides. All fields are optional — omitted\n * values fall back to the voice's stored defaults.\n * @see https://elevenlabs.io/docs/api-reference/text-to-speech/convert\n */\nexport interface ElevenLabsVoiceSettings {\n /** Voice stability, 0..1. Default 0.5. */\n stability?: number\n /** Similarity boost, 0..1. Default 0.75. */\n similarityBoost?: number\n /** Style exaggeration, 0..1. Default 0. */\n style?: number\n /** Playback speed. Default 1.0. */\n speed?: number\n /** Clarity/presence boost. Default true. */\n useSpeakerBoost?: boolean\n}\n\n/**\n * Provider-specific TTS options. `voice` on `generateSpeech()` takes priority\n * over `voiceId` here, but we expose the same field for callers that prefer\n * to keep voice configuration inside the adapter config.\n */\nexport interface ElevenLabsSpeechProviderOptions {\n /** ElevenLabs voice ID to synthesize. Required if `generateSpeech().voice` is not set. */\n voiceId?: string\n /** Output audio format encoded as `codec_samplerate[_bitrate]`. Defaults to `mp3_44100_128`. */\n outputFormat?: ElevenLabsOutputFormat\n /** Voice-settings overrides for this request only. */\n voiceSettings?: ElevenLabsVoiceSettings\n /** ISO-639-1 language code to enforce (e.g. `'en'`, `'ja'`). */\n languageCode?: string\n /** Deterministic sampling seed, 0..4294967295. */\n seed?: number\n /** Previous text for stitching adjacent clips. */\n previousText?: string\n /** Next text for stitching adjacent clips. */\n nextText?: string\n /** Previous request IDs for stitching (max 3). */\n previousRequestIds?: Array<string>\n /** Next request IDs for stitching (max 3). */\n nextRequestIds?: Array<string>\n /** Text normalization toggle. Default `'auto'`. */\n applyTextNormalization?: 'auto' | 'on' | 'off'\n /** Language-specific text normalization (currently Japanese only, adds latency). */\n applyLanguageTextNormalization?: boolean\n /** Latency optimization level, 0..4. */\n optimizeStreamingLatency?: number\n /** Enable logging. Set false for zero-retention mode (enterprise only). */\n enableLogging?: boolean\n}\n\n/**\n * ElevenLabs text-to-speech adapter built on the official\n * `@elevenlabs/elevenlabs-js` SDK.\n *\n * @example\n * ```ts\n * const adapter = elevenlabsSpeech('eleven_multilingual_v2')\n * const result = await generateSpeech({\n * adapter,\n * text: 'Hello, world!',\n * voice: '21m00Tcm4TlvDq8ikWAM',\n * })\n * ```\n */\nexport class ElevenLabsSpeechAdapter<\n TModel extends ElevenLabsTTSModel,\n> extends BaseTTSAdapter<TModel, ElevenLabsSpeechProviderOptions> {\n readonly name = 'elevenlabs' as const\n\n private readonly client: ElevenLabsClient\n\n constructor(model: TModel, config?: ElevenLabsClientConfig) {\n super(model, config ?? {})\n this.client = createElevenLabsClient(config)\n }\n\n async generateSpeech(\n options: TTSOptions<ElevenLabsSpeechProviderOptions>,\n ): Promise<TTSResult> {\n const { logger } = options\n logger.request(\n `activity=generateSpeech provider=elevenlabs model=${this.model}`,\n { provider: 'elevenlabs', model: this.model },\n )\n try {\n const voiceId = options.voice ?? options.modelOptions?.voiceId\n if (!voiceId) {\n throw new Error(\n 'ElevenLabs TTS requires a voice. Pass `voice` on generateSpeech() or `voiceId` in modelOptions.',\n )\n }\n const {\n outputFormat,\n voiceSettings,\n languageCode,\n seed,\n previousText,\n nextText,\n previousRequestIds,\n nextRequestIds,\n applyTextNormalization,\n applyLanguageTextNormalization,\n optimizeStreamingLatency,\n enableLogging,\n } = options.modelOptions ?? {}\n const effectiveOutputFormat =\n outputFormat ?? inferOutputFormatFromResponseFormat(options.format)\n\n const stream = await this.client.textToSpeech.convert(voiceId, {\n text: options.text,\n modelId: this.model,\n ...(effectiveOutputFormat\n ? { outputFormat: effectiveOutputFormat }\n : {}),\n ...(voiceSettings\n ? { voiceSettings: mapVoiceSettings(voiceSettings, options.speed) }\n : options.speed != null\n ? { voiceSettings: { speed: options.speed } }\n : {}),\n ...(languageCode ? { languageCode } : {}),\n ...(seed != null ? { seed } : {}),\n ...(previousText ? { previousText } : {}),\n ...(nextText ? { nextText } : {}),\n ...(previousRequestIds ? { previousRequestIds } : {}),\n ...(nextRequestIds ? { nextRequestIds } : {}),\n ...(applyTextNormalization ? { applyTextNormalization } : {}),\n ...(applyLanguageTextNormalization != null\n ? { applyLanguageTextNormalization }\n : {}),\n ...(optimizeStreamingLatency != null\n ? { optimizeStreamingLatency }\n : {}),\n ...(enableLogging != null ? { enableLogging } : {}),\n })\n\n const buffer = await readStreamToArrayBuffer(stream)\n const base64 = arrayBufferToBase64(buffer)\n const { format, contentType } = parseOutputFormat(effectiveOutputFormat)\n\n return {\n id: generateId(this.name),\n model: this.model,\n audio: base64,\n format,\n contentType,\n }\n } catch (error) {\n logger.errors('elevenlabs.generateSpeech fatal', {\n error,\n source: 'elevenlabs.generateSpeech',\n })\n throw error\n }\n }\n\n protected override generateId(): string {\n return generateId(this.name)\n }\n}\n\nfunction mapVoiceSettings(\n settings: ElevenLabsVoiceSettings,\n speedOverride: number | undefined,\n): Record<string, unknown> {\n return {\n ...(settings.stability != null ? { stability: settings.stability } : {}),\n ...(settings.similarityBoost != null\n ? { similarityBoost: settings.similarityBoost }\n : {}),\n ...(settings.style != null ? { style: settings.style } : {}),\n ...(speedOverride != null\n ? { speed: speedOverride }\n : settings.speed != null\n ? { speed: settings.speed }\n : {}),\n ...(settings.useSpeakerBoost != null\n ? { useSpeakerBoost: settings.useSpeakerBoost }\n : {}),\n }\n}\n\n/**\n * Map the standard TTSOptions `format` (mp3/opus/aac/flac/wav/pcm) to a\n * reasonable ElevenLabs `outputFormat` so callers don't need to know the\n * full codec/samplerate string for the common case.\n */\nfunction inferOutputFormatFromResponseFormat(\n format: TTSOptions['format'] | undefined,\n): ElevenLabsOutputFormat | undefined {\n switch (format) {\n case 'mp3':\n return 'mp3_44100_128'\n case 'pcm':\n return 'pcm_44100'\n case 'opus':\n return 'opus_48000_128'\n case undefined:\n return undefined\n case 'aac':\n case 'flac':\n case 'wav':\n default:\n // `aac` / `flac` / `wav` are not native ElevenLabs formats —\n // fall back to mp3 rather than blowing up mid-request.\n return 'mp3_44100_128'\n }\n}\n\n/**\n * Create an ElevenLabs speech adapter using `ELEVENLABS_API_KEY` from env.\n */\nexport function elevenlabsSpeech<TModel extends ElevenLabsTTSModel>(\n model: TModel,\n config?: ElevenLabsClientConfig,\n): ElevenLabsSpeechAdapter<TModel> {\n return new ElevenLabsSpeechAdapter(model, config)\n}\n\n/**\n * Create an ElevenLabs speech adapter with an explicit API key.\n */\nexport function createElevenLabsSpeech<TModel extends ElevenLabsTTSModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<ElevenLabsClientConfig, 'apiKey'>,\n): ElevenLabsSpeechAdapter<TModel> {\n return new ElevenLabsSpeechAdapter(model, { apiKey, ...config })\n}\n"],"names":[],"mappings":";;AA+EO,MAAM,gCAEH,eAAwD;AAAA,EACvD,OAAO;AAAA,EAEC;AAAA,EAEjB,YAAY,OAAe,QAAiC;AAC1D,UAAM,OAAO,UAAU,EAAE;AACzB,SAAK,SAAS,uBAAuB,MAAM;AAAA,EAC7C;AAAA,EAEA,MAAM,eACJ,SACoB;AACpB,UAAM,EAAE,WAAW;AACnB,WAAO;AAAA,MACL,qDAAqD,KAAK,KAAK;AAAA,MAC/D,EAAE,UAAU,cAAc,OAAO,KAAK,MAAA;AAAA,IAAM;AAE9C,QAAI;AACF,YAAM,UAAU,QAAQ,SAAS,QAAQ,cAAc;AACvD,UAAI,CAAC,SAAS;AACZ,cAAM,IAAI;AAAA,UACR;AAAA,QAAA;AAAA,MAEJ;AACA,YAAM;AAAA,QACJ;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,QACA;AAAA,MAAA,IACE,QAAQ,gBAAgB,CAAA;AAC5B,YAAM,wBACJ,gBAAgB,oCAAoC,QAAQ,MAAM;AAEpE,YAAM,SAAS,MAAM,KAAK,OAAO,aAAa,QAAQ,SAAS;AAAA,QAC7D,MAAM,QAAQ;AAAA,QACd,SAAS,KAAK;AAAA,QACd,GAAI,wBACA,EAAE,cAAc,sBAAA,IAChB,CAAA;AAAA,QACJ,GAAI,gBACA,EAAE,eAAe,iBAAiB,eAAe,QAAQ,KAAK,MAC9D,QAAQ,SAAS,OACf,EAAE,eAAe,EAAE,OAAO,QAAQ,MAAA,EAAM,IACxC,CAAA;AAAA,QACN,GAAI,eAAe,EAAE,aAAA,IAAiB,CAAA;AAAA,QACtC,GAAI,QAAQ,OAAO,EAAE,KAAA,IAAS,CAAA;AAAA,QAC9B,GAAI,eAAe,EAAE,aAAA,IAAiB,CAAA;AAAA,QACtC,GAAI,WAAW,EAAE,SAAA,IAAa,CAAA;AAAA,QAC9B,GAAI,qBAAqB,EAAE,mBAAA,IAAuB,CAAA;AAAA,QAClD,GAAI,iBAAiB,EAAE,eAAA,IAAmB,CAAA;AAAA,QAC1C,GAAI,yBAAyB,EAAE,uBAAA,IAA2B,CAAA;AAAA,QAC1D,GAAI,kCAAkC,OAClC,EAAE,+BAAA,IACF,CAAA;AAAA,QACJ,GAAI,4BAA4B,OAC5B,EAAE,yBAAA,IACF,CAAA;AAAA,QACJ,GAAI,iBAAiB,OAAO,EAAE,kBAAkB,CAAA;AAAA,MAAC,CAClD;AAED,YAAM,SAAS,MAAM,wBAAwB,MAAM;AACnD,YAAM,SAAS,oBAAoB,MAAM;AACzC,YAAM,EAAE,QAAQ,gBAAgB,kBAAkB,qBAAqB;AAEvE,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB,OAAO,KAAK;AAAA,QACZ,OAAO;AAAA,QACP;AAAA,QACA;AAAA,MAAA;AAAA,IAEJ,SAAS,OAAO;AACd,aAAO,OAAO,mCAAmC;AAAA,QAC/C;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AAAA,EAEmB,aAAqB;AACtC,WAAO,WAAW,KAAK,IAAI;AAAA,EAC7B;AACF;AAEA,SAAS,iBACP,UACA,eACyB;AACzB,SAAO;AAAA,IACL,GAAI,SAAS,aAAa,OAAO,EAAE,WAAW,SAAS,UAAA,IAAc,CAAA;AAAA,IACrE,GAAI,SAAS,mBAAmB,OAC5B,EAAE,iBAAiB,SAAS,gBAAA,IAC5B,CAAA;AAAA,IACJ,GAAI,SAAS,SAAS,OAAO,EAAE,OAAO,SAAS,MAAA,IAAU,CAAA;AAAA,IACzD,GAAI,iBAAiB,OACjB,EAAE,OAAO,cAAA,IACT,SAAS,SAAS,OAChB,EAAE,OAAO,SAAS,MAAA,IAClB,CAAA;AAAA,IACN,GAAI,SAAS,mBAAmB,OAC5B,EAAE,iBAAiB,SAAS,oBAC5B,CAAA;AAAA,EAAC;AAET;AAOA,SAAS,oCACP,QACoC;AACpC,UAAQ,QAAA;AAAA,IACN,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AAAA,IACL;AAGE,aAAO;AAAA,EAAA;AAEb;AAKO,SAAS,iBACd,OACA,QACiC;AACjC,SAAO,IAAI,wBAAwB,OAAO,MAAM;AAClD;AAKO,SAAS,uBACd,OACA,QACA,QACiC;AACjC,SAAO,IAAI,wBAAwB,OAAO,EAAE,QAAQ,GAAG,QAAQ;AACjE;"}
|
|
@@ -1,79 +0,0 @@
|
|
|
1
|
-
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters';
|
|
2
|
-
import { TranscriptionOptions, TranscriptionResult } from '@tanstack/ai';
|
|
3
|
-
import { ElevenLabsClientConfig } from '../utils/client.js';
|
|
4
|
-
import { ElevenLabsTranscriptionModel } from '../model-meta.js';
|
|
5
|
-
/**
|
|
6
|
-
* Provider-specific options for ElevenLabs Scribe transcription. Fields map
|
|
7
|
-
* 1:1 onto the SDK's `BodySpeechToTextV1SpeechToTextPost` — mirroring the
|
|
8
|
-
* names so documentation stays useful.
|
|
9
|
-
* @see https://elevenlabs.io/docs/api-reference/speech-to-text/convert
|
|
10
|
-
*/
|
|
11
|
-
export interface ElevenLabsTranscriptionProviderOptions {
|
|
12
|
-
/** Annotate non-speech events like (laughter), (footsteps), …. */
|
|
13
|
-
tagAudioEvents?: boolean;
|
|
14
|
-
/** Maximum number of speakers in the audio (1..32). */
|
|
15
|
-
numSpeakers?: number;
|
|
16
|
-
/** Timestamp granularity for words. */
|
|
17
|
-
timestampsGranularity?: 'word' | 'character' | 'none';
|
|
18
|
-
/** Enable speaker diarization. */
|
|
19
|
-
diarize?: boolean;
|
|
20
|
-
/** Diarization threshold (requires `diarize=true` and no `numSpeakers`). */
|
|
21
|
-
diarizationThreshold?: number;
|
|
22
|
-
/** Detect speaker roles (agent/customer). Requires diarize=true. */
|
|
23
|
-
detectSpeakerRoles?: boolean;
|
|
24
|
-
/** Bias the model towards these keyterms (max 1000). */
|
|
25
|
-
keyterms?: Array<string>;
|
|
26
|
-
/**
|
|
27
|
-
* Entity detection: `'all'`, a category (`'pii'`, `'phi'`, `'pci'`,
|
|
28
|
-
* `'other'`, `'offensive_language'`), or a specific entity type.
|
|
29
|
-
*/
|
|
30
|
-
entityDetection?: string;
|
|
31
|
-
/** Redact entities from the transcript text. Must be a subset of `entityDetection`. */
|
|
32
|
-
entityRedaction?: string;
|
|
33
|
-
/** How redacted entities are formatted. */
|
|
34
|
-
entityRedactionMode?: string;
|
|
35
|
-
/** Whether to skip filler words / non-speech sounds (scribe_v2 only). */
|
|
36
|
-
noVerbatim?: boolean;
|
|
37
|
-
/** Sampling temperature (0..2). */
|
|
38
|
-
temperature?: number;
|
|
39
|
-
/** Deterministic sampling seed (0..2147483647). */
|
|
40
|
-
seed?: number;
|
|
41
|
-
/** Use `false` for zero-retention mode (enterprise only). */
|
|
42
|
-
enableLogging?: boolean;
|
|
43
|
-
/** Multi-channel audio with one speaker per channel. Max 5 channels. */
|
|
44
|
-
useMultiChannel?: boolean;
|
|
45
|
-
/**
|
|
46
|
-
* Hint for audio format. Use `'pcm_s16le_16'` to skip encoding for 16-bit
|
|
47
|
-
* PCM @ 16kHz mono little-endian inputs (lower latency).
|
|
48
|
-
*/
|
|
49
|
-
fileFormat?: 'pcm_s16le_16' | 'other';
|
|
50
|
-
}
|
|
51
|
-
/**
|
|
52
|
-
* ElevenLabs speech-to-text adapter built on the official SDK's Scribe family.
|
|
53
|
-
*
|
|
54
|
-
* @example
|
|
55
|
-
* ```ts
|
|
56
|
-
* const adapter = elevenlabsTranscription('scribe_v1')
|
|
57
|
-
* const result = await generateTranscription({
|
|
58
|
-
* adapter,
|
|
59
|
-
* audio: fileInput,
|
|
60
|
-
* language: 'en',
|
|
61
|
-
* })
|
|
62
|
-
* ```
|
|
63
|
-
*/
|
|
64
|
-
export declare class ElevenLabsTranscriptionAdapter<TModel extends ElevenLabsTranscriptionModel> extends BaseTranscriptionAdapter<TModel, ElevenLabsTranscriptionProviderOptions> {
|
|
65
|
-
readonly name: "elevenlabs";
|
|
66
|
-
private readonly client;
|
|
67
|
-
constructor(model: TModel, config?: ElevenLabsClientConfig);
|
|
68
|
-
transcribe(options: TranscriptionOptions<ElevenLabsTranscriptionProviderOptions>): Promise<TranscriptionResult>;
|
|
69
|
-
private transformResponse;
|
|
70
|
-
protected generateId(): string;
|
|
71
|
-
}
|
|
72
|
-
/**
|
|
73
|
-
* Create an ElevenLabs transcription adapter using `ELEVENLABS_API_KEY` from env.
|
|
74
|
-
*/
|
|
75
|
-
export declare function elevenlabsTranscription<TModel extends ElevenLabsTranscriptionModel>(model: TModel, config?: ElevenLabsClientConfig): ElevenLabsTranscriptionAdapter<TModel>;
|
|
76
|
-
/**
|
|
77
|
-
* Create an ElevenLabs transcription adapter with an explicit API key.
|
|
78
|
-
*/
|
|
79
|
-
export declare function createElevenLabsTranscription<TModel extends ElevenLabsTranscriptionModel>(model: TModel, apiKey: string, config?: Omit<ElevenLabsClientConfig, 'apiKey'>): ElevenLabsTranscriptionAdapter<TModel>;
|