@tanstack/ai-elevenlabs 0.4.4 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +2 -2
- package/dist/esm/adapters/audio.js +3 -3
- package/dist/esm/adapters/audio.js.map +1 -1
- package/dist/esm/adapters/speech.d.ts +13 -1
- package/dist/esm/adapters/speech.js +137 -14
- package/dist/esm/adapters/speech.js.map +1 -1
- package/dist/esm/adapters/voice.d.ts +85 -0
- package/dist/esm/adapters/voice.js +154 -0
- package/dist/esm/adapters/voice.js.map +1 -0
- package/dist/esm/index.d.ts +2 -1
- package/dist/esm/index.js +3 -2
- package/dist/esm/model-meta.d.ts +35 -10
- package/dist/esm/model-meta.js +38 -8
- package/dist/esm/model-meta.js.map +1 -1
- package/package.json +5 -5
- package/src/adapters/audio.ts +4 -3
- package/src/adapters/speech.ts +213 -24
- package/src/adapters/voice.ts +299 -0
- package/src/index.ts +13 -0
- package/src/model-meta.ts +61 -13
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
import { arrayBufferToBase64, createElevenLabsClient, generateId, parseOutputFormat } from "../utils/client.js";
|
|
2
|
+
import { BaseVoiceAdapter } from "@tanstack/ai/adapters";
|
|
3
|
+
//#region src/adapters/voice.ts
|
|
4
|
+
/**
|
|
5
|
+
* ElevenLabs voice-design adapter built on the official
|
|
6
|
+
* `@elevenlabs/elevenlabs-js` SDK.
|
|
7
|
+
*
|
|
8
|
+
* ElevenLabs designs voices in two steps — generate previews, then promote one
|
|
9
|
+
* into a real voice — but that is an implementation detail. Pass `name` to get
|
|
10
|
+
* a saved voice back; leave it off to audition previews first.
|
|
11
|
+
*
|
|
12
|
+
* @example Audition previews
|
|
13
|
+
* ```ts
|
|
14
|
+
* const result = await generateVoice({
|
|
15
|
+
* adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
|
|
16
|
+
* prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
|
|
17
|
+
* })
|
|
18
|
+
* ```
|
|
19
|
+
*
|
|
20
|
+
* @example Save the voice
|
|
21
|
+
* ```ts
|
|
22
|
+
* const result = await generateVoice({
|
|
23
|
+
* adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
|
|
24
|
+
* prompt: 'A bright, upbeat product demo host',
|
|
25
|
+
* name: 'Demo Host',
|
|
26
|
+
* })
|
|
27
|
+
* ```
|
|
28
|
+
*/
|
|
29
|
+
var ElevenLabsVoiceAdapter = class extends BaseVoiceAdapter {
|
|
30
|
+
name = "elevenlabs";
|
|
31
|
+
client;
|
|
32
|
+
constructor(model, config) {
|
|
33
|
+
super(model, config ?? {});
|
|
34
|
+
this.client = createElevenLabsClient(config);
|
|
35
|
+
}
|
|
36
|
+
async generateVoice(options) {
|
|
37
|
+
const { logger } = options;
|
|
38
|
+
logger.request(`activity=generateVoice provider=elevenlabs model=${this.model}`, {
|
|
39
|
+
provider: "elevenlabs",
|
|
40
|
+
model: this.model
|
|
41
|
+
});
|
|
42
|
+
try {
|
|
43
|
+
if (!options.prompt) throw new Error("ElevenLabs voice design requires a `prompt` describing the voice. Reference audio alone is not supported; pass both to guide the design with a real speaker.");
|
|
44
|
+
const opts = options.modelOptions ?? {};
|
|
45
|
+
const referenceAudioBase64 = await toReferenceAudioBase64(options.referenceAudio);
|
|
46
|
+
if (referenceAudioBase64 && this.model !== "eleven_ttv_v3") throw new Error(`ElevenLabs only accepts reference audio on eleven_ttv_v3, but this adapter is using "${this.model}".`);
|
|
47
|
+
const requestOptions = options.abortSignal ? { abortSignal: options.abortSignal } : {};
|
|
48
|
+
const previewResponse = await this.client.textToVoice.design({
|
|
49
|
+
voiceDescription: options.prompt,
|
|
50
|
+
modelId: this.model,
|
|
51
|
+
...opts.outputFormat ? { outputFormat: opts.outputFormat } : {},
|
|
52
|
+
...opts.text ? { text: opts.text } : {},
|
|
53
|
+
...opts.autoGenerateText != null ? { autoGenerateText: opts.autoGenerateText } : {},
|
|
54
|
+
...opts.loudness != null ? { loudness: opts.loudness } : {},
|
|
55
|
+
...opts.seed != null ? { seed: opts.seed } : {},
|
|
56
|
+
...opts.guidanceScale != null ? { guidanceScale: opts.guidanceScale } : {},
|
|
57
|
+
...opts.quality != null ? { quality: opts.quality } : {},
|
|
58
|
+
...opts.shouldEnhance != null ? { shouldEnhance: opts.shouldEnhance } : {},
|
|
59
|
+
...opts.promptStrength != null ? { promptStrength: opts.promptStrength } : {},
|
|
60
|
+
...referenceAudioBase64 ? { referenceAudioBase64 } : {},
|
|
61
|
+
...opts.remixingSessionId ? { remixingSessionId: opts.remixingSessionId } : {},
|
|
62
|
+
...opts.remixingSessionIterationId ? { remixingSessionIterationId: opts.remixingSessionIterationId } : {}
|
|
63
|
+
}, requestOptions);
|
|
64
|
+
const { format, contentType } = parseOutputFormat(opts.outputFormat);
|
|
65
|
+
const voices = previewResponse.previews.map((preview) => ({
|
|
66
|
+
voiceId: preview.generatedVoiceId,
|
|
67
|
+
audio: preview.audioBase64,
|
|
68
|
+
format,
|
|
69
|
+
contentType: preview.mediaType || contentType,
|
|
70
|
+
duration: preview.durationSecs,
|
|
71
|
+
...preview.language ? { language: preview.language } : {},
|
|
72
|
+
saved: false,
|
|
73
|
+
status: "ready"
|
|
74
|
+
}));
|
|
75
|
+
if (voices.length === 0) throw new Error("ElevenLabs returned no voice previews for this description. Try a longer, more specific prompt.");
|
|
76
|
+
if (options.name) voices[0] = await this.promoteFirstPreview(voices, options.name, options.description ?? options.prompt, opts.labels, requestOptions);
|
|
77
|
+
return {
|
|
78
|
+
id: generateId(this.name),
|
|
79
|
+
model: this.model,
|
|
80
|
+
voices,
|
|
81
|
+
previewText: previewResponse.text
|
|
82
|
+
};
|
|
83
|
+
} catch (error) {
|
|
84
|
+
logger.errors("elevenlabs.generateVoice fatal", {
|
|
85
|
+
error,
|
|
86
|
+
source: "elevenlabs.generateVoice"
|
|
87
|
+
});
|
|
88
|
+
throw error;
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Promote the best preview into a real library voice. The remaining
|
|
93
|
+
* generated ids go along as `playedNotSelectedVoiceIds` — ElevenLabs uses
|
|
94
|
+
* them as RLHF signal for future designs.
|
|
95
|
+
*/
|
|
96
|
+
async promoteFirstPreview(voices, voiceName, voiceDescription, labels, requestOptions) {
|
|
97
|
+
const [best, ...rest] = voices;
|
|
98
|
+
if (!best) throw new Error("No preview to promote into a library voice.");
|
|
99
|
+
const saved = await this.client.textToVoice.create({
|
|
100
|
+
voiceName,
|
|
101
|
+
voiceDescription,
|
|
102
|
+
generatedVoiceId: best.voiceId,
|
|
103
|
+
...labels ? { labels } : {},
|
|
104
|
+
...rest.length > 0 ? { playedNotSelectedVoiceIds: rest.map((voice) => voice.voiceId) } : {}
|
|
105
|
+
}, requestOptions);
|
|
106
|
+
return {
|
|
107
|
+
...best,
|
|
108
|
+
voiceId: saved.voiceId,
|
|
109
|
+
saved: true
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
generateId() {
|
|
113
|
+
return generateId(this.name);
|
|
114
|
+
}
|
|
115
|
+
};
|
|
116
|
+
/**
|
|
117
|
+
* Normalize reference audio to the bare base64 the design endpoint wants.
|
|
118
|
+
*
|
|
119
|
+
* https URLs are rejected rather than fetched: ElevenLabs has no URL field
|
|
120
|
+
* here, and downloading caller-supplied media into memory to inline it is a
|
|
121
|
+
* footgun on large files.
|
|
122
|
+
*/
|
|
123
|
+
async function toReferenceAudioBase64(audio) {
|
|
124
|
+
if (audio == null) return void 0;
|
|
125
|
+
if (audio instanceof ArrayBuffer) return arrayBufferToBase64(audio);
|
|
126
|
+
if (typeof audio !== "string") return arrayBufferToBase64(await audio.arrayBuffer());
|
|
127
|
+
if (audio.startsWith("data:")) {
|
|
128
|
+
const commaIndex = audio.indexOf(",");
|
|
129
|
+
const header = commaIndex === -1 ? "" : audio.slice(5, commaIndex);
|
|
130
|
+
if (commaIndex === -1 || !/;base64$/i.test(header)) throw new Error("ElevenLabs voice design needs base64 reference audio. Pass a base64 data URL, a base64 string, a Blob, or an ArrayBuffer.");
|
|
131
|
+
return audio.slice(commaIndex + 1);
|
|
132
|
+
}
|
|
133
|
+
if (/^https?:\/\//i.test(audio)) throw new Error("ElevenLabs voice design does not accept reference audio URLs. Read the file yourself and pass a Blob, ArrayBuffer, or base64 string.");
|
|
134
|
+
return audio;
|
|
135
|
+
}
|
|
136
|
+
/**
|
|
137
|
+
* Create an ElevenLabs voice-design adapter using `ELEVENLABS_API_KEY` from env.
|
|
138
|
+
*/
|
|
139
|
+
function elevenlabsVoiceDesign(model, config) {
|
|
140
|
+
return new ElevenLabsVoiceAdapter(model, config);
|
|
141
|
+
}
|
|
142
|
+
/**
|
|
143
|
+
* Create an ElevenLabs voice-design adapter with an explicit API key.
|
|
144
|
+
*/
|
|
145
|
+
function createElevenLabsVoiceDesign(model, apiKey, config) {
|
|
146
|
+
return new ElevenLabsVoiceAdapter(model, {
|
|
147
|
+
apiKey,
|
|
148
|
+
...config
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
//#endregion
|
|
152
|
+
export { ElevenLabsVoiceAdapter, createElevenLabsVoiceDesign, elevenlabsVoiceDesign };
|
|
153
|
+
|
|
154
|
+
//# sourceMappingURL=voice.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"voice.js","names":[],"sources":["../../../src/adapters/voice.ts"],"sourcesContent":["import { BaseVoiceAdapter } from '@tanstack/ai/adapters'\nimport {\n arrayBufferToBase64,\n createElevenLabsClient,\n generateId,\n parseOutputFormat,\n} from '../utils/client'\nimport type { ElevenLabsClient } from '@elevenlabs/elevenlabs-js'\nimport type {\n GeneratedVoice,\n VoiceGenerationOptions,\n VoiceResult,\n} from '@tanstack/ai'\nimport type { ElevenLabsClientConfig } from '../utils/client'\nimport type {\n ElevenLabsOutputFormat,\n ElevenLabsVoiceModel,\n} from '../model-meta'\n\n/**\n * Provider-specific voice-design options. Fields map 1:1 onto the SDK's\n * `VoiceDesignRequestModel` — mirroring the names so ElevenLabs'\n * documentation stays useful here.\n * @see https://elevenlabs.io/docs/api-reference/text-to-voice/design\n */\nexport interface ElevenLabsVoiceProviderOptions {\n /** Output audio format for the previews, `codec_samplerate[_bitrate]`. */\n outputFormat?: ElevenLabsOutputFormat\n /** Line the previews speak, 100..1000 characters. */\n text?: string\n /** Let ElevenLabs write the preview line from the description. */\n autoGenerateText?: boolean\n /** Preview loudness, -1 (quietest) to 1 (loudest). 0 is roughly -24 LUFS. */\n loudness?: number\n /** Deterministic sampling seed — same seed and inputs produce the same voice. */\n seed?: number\n /** How closely to follow the description. High values can sound robotic. */\n guidanceScale?: number\n /** Higher quality trades variety for fidelity. */\n quality?: number\n /** Let ElevenLabs expand a short description into a detailed one. */\n shouldEnhance?: boolean\n /**\n * Balance of description against reference audio, 0 (almost all reference)\n * to 1 (almost all description). `eleven_ttv_v3` only.\n */\n promptStrength?: number\n /** Metadata stored on the voice. Only used when the voice is saved. */\n labels?: Record<string, string>\n /** Remixing session to attach these generations to. */\n remixingSessionId?: string\n /** Remixing session iteration to attach these generations to. */\n remixingSessionIterationId?: string\n}\n\n/**\n * ElevenLabs voice-design adapter built on the official\n * `@elevenlabs/elevenlabs-js` SDK.\n *\n * ElevenLabs designs voices in two steps — generate previews, then promote one\n * into a real voice — but that is an implementation detail. Pass `name` to get\n * a saved voice back; leave it off to audition previews first.\n *\n * @example Audition previews\n * ```ts\n * const result = await generateVoice({\n * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),\n * prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',\n * })\n * ```\n *\n * @example Save the voice\n * ```ts\n * const result = await generateVoice({\n * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),\n * prompt: 'A bright, upbeat product demo host',\n * name: 'Demo Host',\n * })\n * ```\n */\nexport class ElevenLabsVoiceAdapter<\n TModel extends ElevenLabsVoiceModel,\n> extends BaseVoiceAdapter<TModel, ElevenLabsVoiceProviderOptions> {\n readonly name = 'elevenlabs' as const\n\n private readonly client: ElevenLabsClient\n\n constructor(model: TModel, config?: ElevenLabsClientConfig) {\n super(model, config ?? {})\n this.client = createElevenLabsClient(config)\n }\n\n async generateVoice(\n options: VoiceGenerationOptions<ElevenLabsVoiceProviderOptions>,\n ): Promise<VoiceResult> {\n const { logger } = options\n logger.request(\n `activity=generateVoice provider=elevenlabs model=${this.model}`,\n { provider: 'elevenlabs', model: this.model },\n )\n try {\n // `voiceDescription` is required by the design endpoint, so reference\n // audio alone is not enough here — unlike clone-only providers.\n if (!options.prompt) {\n throw new Error(\n 'ElevenLabs voice design requires a `prompt` describing the voice. Reference audio alone is not supported; pass both to guide the design with a real speaker.',\n )\n }\n\n const opts = options.modelOptions ?? {}\n const referenceAudioBase64 = await toReferenceAudioBase64(\n options.referenceAudio,\n )\n if (referenceAudioBase64 && this.model !== 'eleven_ttv_v3') {\n throw new Error(\n `ElevenLabs only accepts reference audio on eleven_ttv_v3, but this adapter is using \"${this.model}\".`,\n )\n }\n\n const requestOptions = options.abortSignal\n ? { abortSignal: options.abortSignal }\n : {}\n\n const previewResponse = await this.client.textToVoice.design(\n {\n voiceDescription: options.prompt,\n modelId: this.model,\n ...(opts.outputFormat ? { outputFormat: opts.outputFormat } : {}),\n ...(opts.text ? { text: opts.text } : {}),\n ...(opts.autoGenerateText != null\n ? { autoGenerateText: opts.autoGenerateText }\n : {}),\n ...(opts.loudness != null ? { loudness: opts.loudness } : {}),\n ...(opts.seed != null ? { seed: opts.seed } : {}),\n ...(opts.guidanceScale != null\n ? { guidanceScale: opts.guidanceScale }\n : {}),\n ...(opts.quality != null ? { quality: opts.quality } : {}),\n ...(opts.shouldEnhance != null\n ? { shouldEnhance: opts.shouldEnhance }\n : {}),\n ...(opts.promptStrength != null\n ? { promptStrength: opts.promptStrength }\n : {}),\n ...(referenceAudioBase64 ? { referenceAudioBase64 } : {}),\n ...(opts.remixingSessionId\n ? { remixingSessionId: opts.remixingSessionId }\n : {}),\n ...(opts.remixingSessionIterationId\n ? { remixingSessionIterationId: opts.remixingSessionIterationId }\n : {}),\n },\n requestOptions,\n )\n\n const { format, contentType } = parseOutputFormat(opts.outputFormat)\n const voices: Array<GeneratedVoice> = previewResponse.previews.map(\n (preview) => ({\n voiceId: preview.generatedVoiceId,\n audio: preview.audioBase64,\n format,\n contentType: preview.mediaType || contentType,\n duration: preview.durationSecs,\n ...(preview.language ? { language: preview.language } : {}),\n saved: false,\n // The design endpoint returns a finished preview; nothing trains.\n status: 'ready' as const,\n }),\n )\n\n // A caller who passed `name` asked for a persisted voice. Returning an\n // empty list would report success for a request that produced nothing.\n if (voices.length === 0) {\n throw new Error(\n 'ElevenLabs returned no voice previews for this description. Try a longer, more specific prompt.',\n )\n }\n\n if (options.name) {\n voices[0] = await this.promoteFirstPreview(\n voices,\n options.name,\n options.description ?? options.prompt,\n opts.labels,\n requestOptions,\n )\n }\n\n return {\n id: generateId(this.name),\n model: this.model,\n voices,\n previewText: previewResponse.text,\n }\n } catch (error) {\n logger.errors('elevenlabs.generateVoice fatal', {\n error,\n source: 'elevenlabs.generateVoice',\n })\n throw error\n }\n }\n\n /**\n * Promote the best preview into a real library voice. The remaining\n * generated ids go along as `playedNotSelectedVoiceIds` — ElevenLabs uses\n * them as RLHF signal for future designs.\n */\n private async promoteFirstPreview(\n voices: ReadonlyArray<GeneratedVoice>,\n voiceName: string,\n voiceDescription: string,\n labels: Record<string, string> | undefined,\n requestOptions: { abortSignal?: AbortSignal },\n ): Promise<GeneratedVoice> {\n const [best, ...rest] = voices\n if (!best) {\n throw new Error('No preview to promote into a library voice.')\n }\n\n const saved = await this.client.textToVoice.create(\n {\n voiceName,\n voiceDescription,\n generatedVoiceId: best.voiceId,\n ...(labels ? { labels } : {}),\n ...(rest.length > 0\n ? { playedNotSelectedVoiceIds: rest.map((voice) => voice.voiceId) }\n : {}),\n },\n requestOptions,\n )\n\n return { ...best, voiceId: saved.voiceId, saved: true }\n }\n\n protected override generateId(): string {\n return generateId(this.name)\n }\n}\n\n/**\n * Normalize reference audio to the bare base64 the design endpoint wants.\n *\n * https URLs are rejected rather than fetched: ElevenLabs has no URL field\n * here, and downloading caller-supplied media into memory to inline it is a\n * footgun on large files.\n */\nasync function toReferenceAudioBase64(\n audio: VoiceGenerationOptions['referenceAudio'],\n): Promise<string | undefined> {\n if (audio == null) return undefined\n if (audio instanceof ArrayBuffer) return arrayBufferToBase64(audio)\n if (typeof audio !== 'string') {\n return arrayBufferToBase64(await audio.arrayBuffer())\n }\n\n if (audio.startsWith('data:')) {\n const commaIndex = audio.indexOf(',')\n const header = commaIndex === -1 ? '' : audio.slice(5, commaIndex)\n if (commaIndex === -1 || !/;base64$/i.test(header)) {\n throw new Error(\n 'ElevenLabs voice design needs base64 reference audio. Pass a base64 data URL, a base64 string, a Blob, or an ArrayBuffer.',\n )\n }\n return audio.slice(commaIndex + 1)\n }\n\n if (/^https?:\\/\\//i.test(audio)) {\n throw new Error(\n 'ElevenLabs voice design does not accept reference audio URLs. Read the file yourself and pass a Blob, ArrayBuffer, or base64 string.',\n )\n }\n\n return audio\n}\n\n/**\n * Create an ElevenLabs voice-design adapter using `ELEVENLABS_API_KEY` from env.\n */\nexport function elevenlabsVoiceDesign<TModel extends ElevenLabsVoiceModel>(\n model: TModel,\n config?: ElevenLabsClientConfig,\n): ElevenLabsVoiceAdapter<TModel> {\n return new ElevenLabsVoiceAdapter(model, config)\n}\n\n/**\n * Create an ElevenLabs voice-design adapter with an explicit API key.\n */\nexport function createElevenLabsVoiceDesign<\n TModel extends ElevenLabsVoiceModel,\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<ElevenLabsClientConfig, 'apiKey'>,\n): ElevenLabsVoiceAdapter<TModel> {\n return new ElevenLabsVoiceAdapter(model, { apiKey, ...config })\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;AAgFA,IAAa,yBAAb,cAEU,iBAAyD;CACjE,OAAgB;CAEhB;CAEA,YAAY,OAAe,QAAiC;EAC1D,MAAM,OAAO,UAAU,CAAC,CAAC;EACzB,KAAK,SAAS,uBAAuB,MAAM;CAC7C;CAEA,MAAM,cACJ,SACsB;EACtB,MAAM,EAAE,WAAW;EACnB,OAAO,QACL,oDAAoD,KAAK,SACzD;GAAE,UAAU;GAAc,OAAO,KAAK;EAAM,CAC9C;EACA,IAAI;GAGF,IAAI,CAAC,QAAQ,QACX,MAAM,IAAI,MACR,8JACF;GAGF,MAAM,OAAO,QAAQ,gBAAgB,CAAC;GACtC,MAAM,uBAAuB,MAAM,uBACjC,QAAQ,cACV;GACA,IAAI,wBAAwB,KAAK,UAAU,iBACzC,MAAM,IAAI,MACR,wFAAwF,KAAK,MAAM,GACrG;GAGF,MAAM,iBAAiB,QAAQ,cAC3B,EAAE,aAAa,QAAQ,YAAY,IACnC,CAAC;GAEL,MAAM,kBAAkB,MAAM,KAAK,OAAO,YAAY,OACpD;IACE,kBAAkB,QAAQ;IAC1B,SAAS,KAAK;IACd,GAAI,KAAK,eAAe,EAAE,cAAc,KAAK,aAAa,IAAI,CAAC;IAC/D,GAAI,KAAK,OAAO,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;IACvC,GAAI,KAAK,oBAAoB,OACzB,EAAE,kBAAkB,KAAK,iBAAiB,IAC1C,CAAC;IACL,GAAI,KAAK,YAAY,OAAO,EAAE,UAAU,KAAK,SAAS,IAAI,CAAC;IAC3D,GAAI,KAAK,QAAQ,OAAO,EAAE,MAAM,KAAK,KAAK,IAAI,CAAC;IAC/C,GAAI,KAAK,iBAAiB,OACtB,EAAE,eAAe,KAAK,cAAc,IACpC,CAAC;IACL,GAAI,KAAK,WAAW,OAAO,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;IACxD,GAAI,KAAK,iBAAiB,OACtB,EAAE,eAAe,KAAK,cAAc,IACpC,CAAC;IACL,GAAI,KAAK,kBAAkB,OACvB,EAAE,gBAAgB,KAAK,eAAe,IACtC,CAAC;IACL,GAAI,uBAAuB,EAAE,qBAAqB,IAAI,CAAC;IACvD,GAAI,KAAK,oBACL,EAAE,mBAAmB,KAAK,kBAAkB,IAC5C,CAAC;IACL,GAAI,KAAK,6BACL,EAAE,4BAA4B,KAAK,2BAA2B,IAC9D,CAAC;GACP,GACA,cACF;GAEA,MAAM,EAAE,QAAQ,gBAAgB,kBAAkB,KAAK,YAAY;GACnE,MAAM,SAAgC,gBAAgB,SAAS,KAC5D,aAAa;IACZ,SAAS,QAAQ;IACjB,OAAO,QAAQ;IACf;IACA,aAAa,QAAQ,aAAa;IAClC,UAAU,QAAQ;IAClB,GAAI,QAAQ,WAAW,EAAE,UAAU,QAAQ,SAAS,IAAI,CAAC;IACzD,OAAO;IAEP,QAAQ;GACV,EACF;GAIA,IAAI,OAAO,WAAW,GACpB,MAAM,IAAI,MACR,iGACF;GAGF,IAAI,QAAQ,MACV,OAAO,KAAK,MAAM,KAAK,oBACrB,QACA,QAAQ,MACR,QAAQ,eAAe,QAAQ,QAC/B,KAAK,QACL,cACF;GAGF,OAAO;IACL,IAAI,WAAW,KAAK,IAAI;IACxB,OAAO,KAAK;IACZ;IACA,aAAa,gBAAgB;GAC/B;EACF,SAAS,OAAO;GACd,OAAO,OAAO,kCAAkC;IAC9C;IACA,QAAQ;GACV,CAAC;GACD,MAAM;EACR;CACF;;;;;;CAOA,MAAc,oBACZ,QACA,WACA,kBACA,QACA,gBACyB;EACzB,MAAM,CAAC,MAAM,GAAG,QAAQ;EACxB,IAAI,CAAC,MACH,MAAM,IAAI,MAAM,6CAA6C;EAG/D,MAAM,QAAQ,MAAM,KAAK,OAAO,YAAY,OAC1C;GACE;GACA;GACA,kBAAkB,KAAK;GACvB,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;GAC3B,GAAI,KAAK,SAAS,IACd,EAAE,2BAA2B,KAAK,KAAK,UAAU,MAAM,OAAO,EAAE,IAChE,CAAC;EACP,GACA,cACF;EAEA,OAAO;GAAE,GAAG;GAAM,SAAS,MAAM;GAAS,OAAO;EAAK;CACxD;CAEA,aAAwC;EACtC,OAAO,WAAW,KAAK,IAAI;CAC7B;AACF;;;;;;;;AASA,eAAe,uBACb,OAC6B;CAC7B,IAAI,SAAS,MAAM,OAAO,KAAA;CAC1B,IAAI,iBAAiB,aAAa,OAAO,oBAAoB,KAAK;CAClE,IAAI,OAAO,UAAU,UACnB,OAAO,oBAAoB,MAAM,MAAM,YAAY,CAAC;CAGtD,IAAI,MAAM,WAAW,OAAO,GAAG;EAC7B,MAAM,aAAa,MAAM,QAAQ,GAAG;EACpC,MAAM,SAAS,eAAe,KAAK,KAAK,MAAM,MAAM,GAAG,UAAU;EACjE,IAAI,eAAe,MAAM,CAAC,YAAY,KAAK,MAAM,GAC/C,MAAM,IAAI,MACR,2HACF;EAEF,OAAO,MAAM,MAAM,aAAa,CAAC;CACnC;CAEA,IAAI,gBAAgB,KAAK,KAAK,GAC5B,MAAM,IAAI,MACR,sIACF;CAGF,OAAO;AACT;;;;AAKA,SAAgB,sBACd,OACA,QACgC;CAChC,OAAO,IAAI,uBAAuB,OAAO,MAAM;AACjD;;;;AAKA,SAAgB,4BAGd,OACA,QACA,QACgC;CAChC,OAAO,IAAI,uBAAuB,OAAO;EAAE;EAAQ,GAAG;CAAO,CAAC;AAChE"}
|
package/dist/esm/index.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ export { elevenlabsRealtimeToken, elevenlabsRealtime } from './realtime/index.js
|
|
|
2
2
|
export type { ElevenLabsRealtimeTokenOptions, ElevenLabsRealtimeOptions, ElevenLabsConversationMode, ElevenLabsVADConfig, ElevenLabsClientTool, } from './realtime/index.js';
|
|
3
3
|
export { ElevenLabsSpeechAdapter, createElevenLabsSpeech, elevenlabsSpeech, type ElevenLabsSpeechProviderOptions, type ElevenLabsVoiceSettings, } from './adapters/speech.js';
|
|
4
4
|
export { ElevenLabsAudioAdapter, createElevenLabsAudio, elevenlabsAudio, type ElevenLabsAudioProviderOptions, type ElevenLabsMusicProviderOptions, type ElevenLabsSoundEffectsProviderOptions, type ElevenLabsMusicCompositionPlan, } from './adapters/audio.js';
|
|
5
|
+
export { ElevenLabsVoiceAdapter, createElevenLabsVoiceDesign, elevenlabsVoiceDesign, type ElevenLabsVoiceProviderOptions, } from './adapters/voice.js';
|
|
5
6
|
export { ElevenLabsTranscriptionAdapter, createElevenLabsTranscription, elevenlabsTranscription, type ElevenLabsTranscriptionProviderOptions, } from './adapters/transcription.js';
|
|
6
|
-
export { ELEVENLABS_TTS_MODELS, ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel, type ElevenLabsTTSModel, type ElevenLabsAudioModel, type ElevenLabsMusicModel, type ElevenLabsSoundEffectsModel, type ElevenLabsTranscriptionModel, type ElevenLabsOutputFormat, } from './model-meta.js';
|
|
7
|
+
export { ELEVENLABS_TTS_MODELS, ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, ELEVENLABS_VOICE_MODELS, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel, type ElevenLabsTTSModel, type ElevenLabsAudioModel, type ElevenLabsMusicModel, type ElevenLabsSoundEffectsModel, type ElevenLabsTranscriptionModel, type ElevenLabsVoiceModel, type ElevenLabsOutputFormat, } from './model-meta.js';
|
|
7
8
|
export { getElevenLabsApiKeyFromEnv, type ElevenLabsClientConfig, } from './utils/index.js';
|
package/dist/esm/index.js
CHANGED
|
@@ -3,8 +3,9 @@ import { elevenlabsRealtimeToken } from "./realtime/token.js";
|
|
|
3
3
|
import { elevenlabsRealtime } from "./realtime/adapter.js";
|
|
4
4
|
import "./realtime/index.js";
|
|
5
5
|
import { ElevenLabsSpeechAdapter, createElevenLabsSpeech, elevenlabsSpeech } from "./adapters/speech.js";
|
|
6
|
-
import { ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, ELEVENLABS_TTS_MODELS, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel } from "./model-meta.js";
|
|
6
|
+
import { ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, ELEVENLABS_TTS_MODELS, ELEVENLABS_VOICE_MODELS, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel } from "./model-meta.js";
|
|
7
7
|
import { ElevenLabsAudioAdapter, createElevenLabsAudio, elevenlabsAudio } from "./adapters/audio.js";
|
|
8
|
+
import { ElevenLabsVoiceAdapter, createElevenLabsVoiceDesign, elevenlabsVoiceDesign } from "./adapters/voice.js";
|
|
8
9
|
import { ElevenLabsTranscriptionAdapter, createElevenLabsTranscription, elevenlabsTranscription } from "./adapters/transcription.js";
|
|
9
10
|
import "./utils/index.js";
|
|
10
|
-
export { ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, ELEVENLABS_TTS_MODELS, ElevenLabsAudioAdapter, ElevenLabsSpeechAdapter, ElevenLabsTranscriptionAdapter, createElevenLabsAudio, createElevenLabsSpeech, createElevenLabsTranscription, elevenlabsAudio, elevenlabsRealtime, elevenlabsRealtimeToken, elevenlabsSpeech, elevenlabsTranscription, getElevenLabsApiKeyFromEnv, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel };
|
|
11
|
+
export { ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, ELEVENLABS_TTS_MODELS, ELEVENLABS_VOICE_MODELS, ElevenLabsAudioAdapter, ElevenLabsSpeechAdapter, ElevenLabsTranscriptionAdapter, ElevenLabsVoiceAdapter, createElevenLabsAudio, createElevenLabsSpeech, createElevenLabsTranscription, createElevenLabsVoiceDesign, elevenlabsAudio, elevenlabsRealtime, elevenlabsRealtimeToken, elevenlabsSpeech, elevenlabsTranscription, elevenlabsVoiceDesign, getElevenLabsApiKeyFromEnv, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel };
|
package/dist/esm/model-meta.d.ts
CHANGED
|
@@ -1,37 +1,62 @@
|
|
|
1
1
|
import { ElevenLabs } from '@elevenlabs/elevenlabs-js';
|
|
2
2
|
/**
|
|
3
3
|
* ElevenLabs model identifiers. The lists below are the source of truth —
|
|
4
|
-
* callers are blocked from passing unknown model IDs.
|
|
5
|
-
*
|
|
4
|
+
* callers are blocked from passing unknown model IDs.
|
|
5
|
+
*
|
|
6
|
+
* Maintained by hand: `scripts/sync-provider-models.ts` deliberately excludes
|
|
7
|
+
* elevenlabs, because these are media endpoint ids rather than OpenRouter
|
|
8
|
+
* models. Where the SDK publishes its own union, pin the list to it with
|
|
9
|
+
* `satisfies` so a removed id fails the build instead of a request.
|
|
10
|
+
*
|
|
11
|
+
* Each list is ordered newest-first, with models ElevenLabs has deprecated
|
|
12
|
+
* kept at the bottom so existing callers don't break on an upgrade. An id the
|
|
13
|
+
* SDK has dropped outright is removed rather than kept, because the adapter
|
|
14
|
+
* can no longer send it — see `eleven_text_to_sound_v1`.
|
|
15
|
+
*
|
|
16
|
+
* @see https://elevenlabs.io/docs/models
|
|
6
17
|
*/
|
|
7
18
|
/**
|
|
8
19
|
* Text-to-speech models.
|
|
9
20
|
* @see https://elevenlabs.io/docs/models
|
|
10
21
|
*/
|
|
11
|
-
export declare const ELEVENLABS_TTS_MODELS: readonly ["eleven_v3", "eleven_multilingual_v2", "eleven_flash_v2_5", "eleven_flash_v2", "eleven_turbo_v2_5", "eleven_turbo_v2", "eleven_monolingual_v1"];
|
|
22
|
+
export declare const ELEVENLABS_TTS_MODELS: readonly ["eleven_v3", "eleven_v3_conversational", "eleven_multilingual_v2", "eleven_flash_v2_5", "eleven_flash_v2", "eleven_turbo_v2_5", "eleven_turbo_v2", "eleven_monolingual_v1"];
|
|
12
23
|
export type ElevenLabsTTSModel = (typeof ELEVENLABS_TTS_MODELS)[number];
|
|
13
24
|
/**
|
|
14
|
-
* Audio generation models — music (`
|
|
25
|
+
* Audio generation models — music (`music_v*`) + sound effects
|
|
15
26
|
* (`eleven_text_to_sound_v*`) share one `generateAudio` adapter.
|
|
16
27
|
* The adapter dispatches by model id so callers pick behavior via the model.
|
|
17
28
|
*
|
|
18
29
|
* @see https://elevenlabs.io/docs/overview/capabilities/music
|
|
19
30
|
* @see https://elevenlabs.io/docs/overview/capabilities/sound-effects
|
|
20
31
|
*/
|
|
21
|
-
export declare const ELEVENLABS_AUDIO_MODELS: readonly ["
|
|
32
|
+
export declare const ELEVENLABS_AUDIO_MODELS: readonly ["music_v2_5", "music_v2", "eleven_text_to_sound_v2", "music_v1"];
|
|
22
33
|
export type ElevenLabsAudioModel = (typeof ELEVENLABS_AUDIO_MODELS)[number];
|
|
23
|
-
/** Music models within the audio family. */
|
|
24
|
-
export type ElevenLabsMusicModel =
|
|
25
|
-
/** SFX models within the audio family. */
|
|
26
|
-
export type ElevenLabsSoundEffectsModel =
|
|
34
|
+
/** Music models within the audio family, derived from the audio list. */
|
|
35
|
+
export type ElevenLabsMusicModel = Extract<ElevenLabsAudioModel, `music_${string}`>;
|
|
36
|
+
/** SFX models within the audio family, derived from the audio list. */
|
|
37
|
+
export type ElevenLabsSoundEffectsModel = Extract<ElevenLabsAudioModel, `eleven_text_to_sound_${string}`>;
|
|
38
|
+
/**
|
|
39
|
+
* Matches the `music_` prefix so a new music release needs no branch here,
|
|
40
|
+
* but still checks membership: the narrowed type is the closed set above, so
|
|
41
|
+
* a `music_*` id this package does not ship must not pass.
|
|
42
|
+
*/
|
|
27
43
|
export declare function isElevenLabsMusicModel(model: string): model is ElevenLabsMusicModel;
|
|
28
44
|
export declare function isElevenLabsSoundEffectsModel(model: string): model is ElevenLabsSoundEffectsModel;
|
|
29
45
|
/**
|
|
30
46
|
* Speech-to-text (transcription) models — Scribe family.
|
|
31
47
|
* @see https://elevenlabs.io/docs/overview/capabilities/speech-to-text
|
|
32
48
|
*/
|
|
33
|
-
export declare const ELEVENLABS_TRANSCRIPTION_MODELS: readonly ["scribe_v2", "scribe_v1"];
|
|
49
|
+
export declare const ELEVENLABS_TRANSCRIPTION_MODELS: readonly ["scribe_v2", "scribe_v2_medical", "scribe_v1"];
|
|
34
50
|
export type ElevenLabsTranscriptionModel = (typeof ELEVENLABS_TRANSCRIPTION_MODELS)[number];
|
|
51
|
+
/**
|
|
52
|
+
* Voice design (text-to-voice) models, used by the `generateVoice` adapter.
|
|
53
|
+
* `eleven_ttv_v3` is the only one that accepts reference audio, so guiding a
|
|
54
|
+
* design with a recording of a real speaker requires it.
|
|
55
|
+
*
|
|
56
|
+
* @see https://elevenlabs.io/docs/overview/capabilities/voice-design
|
|
57
|
+
*/
|
|
58
|
+
export declare const ELEVENLABS_VOICE_MODELS: readonly ["eleven_ttv_v3", "eleven_multilingual_ttv_v2"];
|
|
59
|
+
export type ElevenLabsVoiceModel = (typeof ELEVENLABS_VOICE_MODELS)[number];
|
|
35
60
|
/**
|
|
36
61
|
* Supported `output_format` strings, encoded as `codec_samplerate[_bitrate]`.
|
|
37
62
|
* Aliased to the SDK's `AllowedOutputFormats` so the list stays in sync
|
package/dist/esm/model-meta.js
CHANGED
|
@@ -1,8 +1,19 @@
|
|
|
1
1
|
//#region src/model-meta.ts
|
|
2
2
|
/**
|
|
3
3
|
* ElevenLabs model identifiers. The lists below are the source of truth —
|
|
4
|
-
* callers are blocked from passing unknown model IDs.
|
|
5
|
-
*
|
|
4
|
+
* callers are blocked from passing unknown model IDs.
|
|
5
|
+
*
|
|
6
|
+
* Maintained by hand: `scripts/sync-provider-models.ts` deliberately excludes
|
|
7
|
+
* elevenlabs, because these are media endpoint ids rather than OpenRouter
|
|
8
|
+
* models. Where the SDK publishes its own union, pin the list to it with
|
|
9
|
+
* `satisfies` so a removed id fails the build instead of a request.
|
|
10
|
+
*
|
|
11
|
+
* Each list is ordered newest-first, with models ElevenLabs has deprecated
|
|
12
|
+
* kept at the bottom so existing callers don't break on an upgrade. An id the
|
|
13
|
+
* SDK has dropped outright is removed rather than kept, because the adapter
|
|
14
|
+
* can no longer send it — see `eleven_text_to_sound_v1`.
|
|
15
|
+
*
|
|
16
|
+
* @see https://elevenlabs.io/docs/models
|
|
6
17
|
*/
|
|
7
18
|
/**
|
|
8
19
|
* Text-to-speech models.
|
|
@@ -10,6 +21,7 @@
|
|
|
10
21
|
*/
|
|
11
22
|
var ELEVENLABS_TTS_MODELS = [
|
|
12
23
|
"eleven_v3",
|
|
24
|
+
"eleven_v3_conversational",
|
|
13
25
|
"eleven_multilingual_v2",
|
|
14
26
|
"eleven_flash_v2_5",
|
|
15
27
|
"eleven_flash_v2",
|
|
@@ -18,7 +30,7 @@ var ELEVENLABS_TTS_MODELS = [
|
|
|
18
30
|
"eleven_monolingual_v1"
|
|
19
31
|
];
|
|
20
32
|
/**
|
|
21
|
-
* Audio generation models — music (`
|
|
33
|
+
* Audio generation models — music (`music_v*`) + sound effects
|
|
22
34
|
* (`eleven_text_to_sound_v*`) share one `generateAudio` adapter.
|
|
23
35
|
* The adapter dispatches by model id so callers pick behavior via the model.
|
|
24
36
|
*
|
|
@@ -26,12 +38,18 @@ var ELEVENLABS_TTS_MODELS = [
|
|
|
26
38
|
* @see https://elevenlabs.io/docs/overview/capabilities/sound-effects
|
|
27
39
|
*/
|
|
28
40
|
var ELEVENLABS_AUDIO_MODELS = [
|
|
29
|
-
"
|
|
41
|
+
"music_v2_5",
|
|
42
|
+
"music_v2",
|
|
30
43
|
"eleven_text_to_sound_v2",
|
|
31
|
-
"
|
|
44
|
+
"music_v1"
|
|
32
45
|
];
|
|
46
|
+
/**
|
|
47
|
+
* Matches the `music_` prefix so a new music release needs no branch here,
|
|
48
|
+
* but still checks membership: the narrowed type is the closed set above, so
|
|
49
|
+
* a `music_*` id this package does not ship must not pass.
|
|
50
|
+
*/
|
|
33
51
|
function isElevenLabsMusicModel(model) {
|
|
34
|
-
return model
|
|
52
|
+
return model.startsWith("music_") && ELEVENLABS_AUDIO_MODELS.includes(model);
|
|
35
53
|
}
|
|
36
54
|
function isElevenLabsSoundEffectsModel(model) {
|
|
37
55
|
return model.startsWith("eleven_text_to_sound_");
|
|
@@ -40,8 +58,20 @@ function isElevenLabsSoundEffectsModel(model) {
|
|
|
40
58
|
* Speech-to-text (transcription) models — Scribe family.
|
|
41
59
|
* @see https://elevenlabs.io/docs/overview/capabilities/speech-to-text
|
|
42
60
|
*/
|
|
43
|
-
var ELEVENLABS_TRANSCRIPTION_MODELS = [
|
|
61
|
+
var ELEVENLABS_TRANSCRIPTION_MODELS = [
|
|
62
|
+
"scribe_v2",
|
|
63
|
+
"scribe_v2_medical",
|
|
64
|
+
"scribe_v1"
|
|
65
|
+
];
|
|
66
|
+
/**
|
|
67
|
+
* Voice design (text-to-voice) models, used by the `generateVoice` adapter.
|
|
68
|
+
* `eleven_ttv_v3` is the only one that accepts reference audio, so guiding a
|
|
69
|
+
* design with a recording of a real speaker requires it.
|
|
70
|
+
*
|
|
71
|
+
* @see https://elevenlabs.io/docs/overview/capabilities/voice-design
|
|
72
|
+
*/
|
|
73
|
+
var ELEVENLABS_VOICE_MODELS = ["eleven_ttv_v3", "eleven_multilingual_ttv_v2"];
|
|
44
74
|
//#endregion
|
|
45
|
-
export { ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, ELEVENLABS_TTS_MODELS, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel };
|
|
75
|
+
export { ELEVENLABS_AUDIO_MODELS, ELEVENLABS_TRANSCRIPTION_MODELS, ELEVENLABS_TTS_MODELS, ELEVENLABS_VOICE_MODELS, isElevenLabsMusicModel, isElevenLabsSoundEffectsModel };
|
|
46
76
|
|
|
47
77
|
//# sourceMappingURL=model-meta.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"model-meta.js","names":[],"sources":["../../src/model-meta.ts"],"sourcesContent":["import type { ElevenLabs } from '@elevenlabs/elevenlabs-js'\n\n/**\n * ElevenLabs model identifiers. The lists below are the source of truth —\n * callers are blocked from passing unknown model IDs
|
|
1
|
+
{"version":3,"file":"model-meta.js","names":[],"sources":["../../src/model-meta.ts"],"sourcesContent":["import type { ElevenLabs } from '@elevenlabs/elevenlabs-js'\n\n/**\n * ElevenLabs model identifiers. The lists below are the source of truth —\n * callers are blocked from passing unknown model IDs.\n *\n * Maintained by hand: `scripts/sync-provider-models.ts` deliberately excludes\n * elevenlabs, because these are media endpoint ids rather than OpenRouter\n * models. Where the SDK publishes its own union, pin the list to it with\n * `satisfies` so a removed id fails the build instead of a request.\n *\n * Each list is ordered newest-first, with models ElevenLabs has deprecated\n * kept at the bottom so existing callers don't break on an upgrade. An id the\n * SDK has dropped outright is removed rather than kept, because the adapter\n * can no longer send it — see `eleven_text_to_sound_v1`.\n *\n * @see https://elevenlabs.io/docs/models\n */\n\n/**\n * Text-to-speech models.\n * @see https://elevenlabs.io/docs/models\n */\nexport const ELEVENLABS_TTS_MODELS = [\n 'eleven_v3',\n 'eleven_v3_conversational',\n 'eleven_multilingual_v2',\n 'eleven_flash_v2_5',\n 'eleven_flash_v2',\n // Deprecated by ElevenLabs — use `eleven_flash_v2_5`.\n 'eleven_turbo_v2_5',\n // Deprecated by ElevenLabs — use `eleven_flash_v2`.\n 'eleven_turbo_v2',\n // Deprecated by ElevenLabs — use `eleven_multilingual_v2`.\n 'eleven_monolingual_v1',\n] as const\n\nexport type ElevenLabsTTSModel = (typeof ELEVENLABS_TTS_MODELS)[number]\n\n/**\n * Audio generation models — music (`music_v*`) + sound effects\n * (`eleven_text_to_sound_v*`) share one `generateAudio` adapter.\n * The adapter dispatches by model id so callers pick behavior via the model.\n *\n * @see https://elevenlabs.io/docs/overview/capabilities/music\n * @see https://elevenlabs.io/docs/overview/capabilities/sound-effects\n */\nexport const ELEVENLABS_AUDIO_MODELS = [\n 'music_v2_5',\n 'music_v2',\n 'eleven_text_to_sound_v2',\n // Deprecated by ElevenLabs — use `music_v2_5`.\n 'music_v1',\n] as const satisfies ReadonlyArray<\n ElevenLabs.MusicModelId | ElevenLabs.SfxModelId\n>\n\nexport type ElevenLabsAudioModel = (typeof ELEVENLABS_AUDIO_MODELS)[number]\n\n/** Music models within the audio family, derived from the audio list. */\nexport type ElevenLabsMusicModel = Extract<\n ElevenLabsAudioModel,\n `music_${string}`\n>\n/** SFX models within the audio family, derived from the audio list. */\nexport type ElevenLabsSoundEffectsModel = Extract<\n ElevenLabsAudioModel,\n `eleven_text_to_sound_${string}`\n>\n\n/**\n * Matches the `music_` prefix so a new music release needs no branch here,\n * but still checks membership: the narrowed type is the closed set above, so\n * a `music_*` id this package does not ship must not pass.\n */\nexport function isElevenLabsMusicModel(\n model: string,\n): model is ElevenLabsMusicModel {\n return (\n model.startsWith('music_') &&\n (ELEVENLABS_AUDIO_MODELS as ReadonlyArray<string>).includes(model)\n )\n}\n\nexport function isElevenLabsSoundEffectsModel(\n model: string,\n): model is ElevenLabsSoundEffectsModel {\n return model.startsWith('eleven_text_to_sound_')\n}\n\n/**\n * Speech-to-text (transcription) models — Scribe family.\n * @see https://elevenlabs.io/docs/overview/capabilities/speech-to-text\n */\nexport const ELEVENLABS_TRANSCRIPTION_MODELS = [\n 'scribe_v2',\n // Domain-tuned for clinical audio.\n 'scribe_v2_medical',\n // Deprecated by ElevenLabs — use `scribe_v2`.\n 'scribe_v1',\n] as const\n\nexport type ElevenLabsTranscriptionModel =\n (typeof ELEVENLABS_TRANSCRIPTION_MODELS)[number]\n\n/**\n * Voice design (text-to-voice) models, used by the `generateVoice` adapter.\n * `eleven_ttv_v3` is the only one that accepts reference audio, so guiding a\n * design with a recording of a real speaker requires it.\n *\n * @see https://elevenlabs.io/docs/overview/capabilities/voice-design\n */\nexport const ELEVENLABS_VOICE_MODELS = [\n 'eleven_ttv_v3',\n 'eleven_multilingual_ttv_v2',\n] as const satisfies ReadonlyArray<ElevenLabs.VoiceDesignRequestModelModelId>\n\nexport type ElevenLabsVoiceModel = (typeof ELEVENLABS_VOICE_MODELS)[number]\n\n/**\n * Supported `output_format` strings, encoded as `codec_samplerate[_bitrate]`.\n * Aliased to the SDK's `AllowedOutputFormats` so the list stays in sync\n * automatically whenever the `@elevenlabs/elevenlabs-js` dependency is bumped.\n *\n * @see https://elevenlabs.io/docs/api-reference/text-to-speech/convert\n */\nexport type ElevenLabsOutputFormat = ElevenLabs.AllowedOutputFormats\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;AAuBA,IAAa,wBAAwB;CACnC;CACA;CACA;CACA;CACA;CAEA;CAEA;CAEA;AACF;;;;;;;;;AAYA,IAAa,0BAA0B;CACrC;CACA;CACA;CAEA;AACF;;;;;;AAsBA,SAAgB,uBACd,OAC+B;CAC/B,OACE,MAAM,WAAW,QAAQ,KACxB,wBAAkD,SAAS,KAAK;AAErE;AAEA,SAAgB,8BACd,OACsC;CACtC,OAAO,MAAM,WAAW,uBAAuB;AACjD;;;;;AAMA,IAAa,kCAAkC;CAC7C;CAEA;CAEA;AACF;;;;;;;;AAYA,IAAa,0BAA0B,CACrC,iBACA,4BACF"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai-elevenlabs",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.6.0",
|
|
4
4
|
"description": "ElevenLabs adapter for TanStack AI realtime voice, text-to-speech, transcription, music, and sound effects.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -52,16 +52,16 @@
|
|
|
52
52
|
"src"
|
|
53
53
|
],
|
|
54
54
|
"dependencies": {
|
|
55
|
-
"@elevenlabs/client": "^1.
|
|
56
|
-
"@elevenlabs/elevenlabs-js": "^2.
|
|
55
|
+
"@elevenlabs/client": "^1.25.0",
|
|
56
|
+
"@elevenlabs/elevenlabs-js": "^2.68.0",
|
|
57
57
|
"@tanstack/ai-utils": "^0.4.0"
|
|
58
58
|
},
|
|
59
59
|
"peerDependencies": {
|
|
60
|
-
"@tanstack/ai": "^0.
|
|
60
|
+
"@tanstack/ai": "^0.57.0"
|
|
61
61
|
},
|
|
62
62
|
"devDependencies": {
|
|
63
63
|
"@vitest/coverage-v8": "4.1.10",
|
|
64
|
-
"@tanstack/ai": "0.
|
|
64
|
+
"@tanstack/ai": "0.57.0"
|
|
65
65
|
},
|
|
66
66
|
"scripts": {
|
|
67
67
|
"build": "vite build",
|
package/src/adapters/audio.ts
CHANGED
|
@@ -7,6 +7,7 @@ import {
|
|
|
7
7
|
readStreamToArrayBuffer,
|
|
8
8
|
} from '../utils/client'
|
|
9
9
|
import {
|
|
10
|
+
ELEVENLABS_AUDIO_MODELS,
|
|
10
11
|
isElevenLabsMusicModel,
|
|
11
12
|
isElevenLabsSoundEffectsModel,
|
|
12
13
|
} from '../model-meta'
|
|
@@ -55,7 +56,7 @@ interface CommonAudioOptions {
|
|
|
55
56
|
}
|
|
56
57
|
|
|
57
58
|
/**
|
|
58
|
-
* Provider options for music generation (`
|
|
59
|
+
* Provider options for music generation (`music_v*`).
|
|
59
60
|
*/
|
|
60
61
|
export interface ElevenLabsMusicProviderOptions extends CommonAudioOptions {
|
|
61
62
|
/** Structured composition plan. Mutually exclusive with `prompt`/`duration`. */
|
|
@@ -94,7 +95,7 @@ export type ElevenLabsAudioProviderOptions =
|
|
|
94
95
|
*
|
|
95
96
|
* @example
|
|
96
97
|
* ```ts
|
|
97
|
-
* const music = elevenlabsAudio('
|
|
98
|
+
* const music = elevenlabsAudio('music_v2_5')
|
|
98
99
|
* await generateAudio({ adapter: music, prompt: 'lo-fi beat', duration: 15 })
|
|
99
100
|
*
|
|
100
101
|
* const sfx = elevenlabsAudio('eleven_text_to_sound_v2')
|
|
@@ -129,7 +130,7 @@ export class ElevenLabsAudioAdapter<
|
|
|
129
130
|
return await this.runSoundEffects(options)
|
|
130
131
|
}
|
|
131
132
|
throw new Error(
|
|
132
|
-
`Unsupported ElevenLabs audio model "${this.model}". Expected one of:
|
|
133
|
+
`Unsupported ElevenLabs audio model "${this.model}". Expected one of: ${ELEVENLABS_AUDIO_MODELS.join(', ')}.`,
|
|
133
134
|
)
|
|
134
135
|
} catch (error) {
|
|
135
136
|
logger.errors('elevenlabs.generateAudio fatal', {
|