dsh-audiogen 0.3.4 → 0.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -250,6 +250,97 @@ async function openAITTS(channel, request, signal) {
250
250
  async function elevenLabs(channel, request, signal) {
251
251
  const base = endpointBase(channel.apiUrl);
252
252
  const model = (request.upstream ?? request.model) || "eleven_multilingual_v2";
253
+ const headers = {
254
+ "xi-api-key": channel.apiKey.trim(),
255
+ authorization: `Bearer ${channel.apiKey.trim()}`,
256
+ "content-type": "application/json",
257
+ accept: "audio/mpeg, application/json"
258
+ };
259
+ if (request.mode === "voice_design") {
260
+ const endpoint = `${base}/text-to-voice/design`;
261
+ const previewText = request.previewText?.trim() ?? "";
262
+ const body = {
263
+ voice_description: request.prompt,
264
+ ...previewText.length >= 100 ? { text: previewText } : { auto_generate_text: true }
265
+ };
266
+ const response = await fetchWithTimeout(endpoint, {
267
+ method: "POST",
268
+ redirect: "error",
269
+ headers,
270
+ body: JSON.stringify(body),
271
+ signal
272
+ }, UPSTREAM_TIMEOUT_MS);
273
+ if (!response.ok) {
274
+ const detail = await response.text().catch(() => "");
275
+ throw new AudioGenError(`ElevenLabs voice design API error (HTTP ${response.status})${detail === "" ? "" : `: ${detail.slice(0, 300)}`}`, "audio-api-error");
276
+ }
277
+ const previews = (await response.json()).previews ?? [];
278
+ if (previews.length === 0) throw new AudioGenError("ElevenLabs voice design returned no previews", "audio-empty-result");
279
+ const outputs = [];
280
+ for (const preview of previews) {
281
+ const encoded = preview.audio_base_64?.trim() ?? "";
282
+ if (encoded === "") continue;
283
+ const data = new Uint8Array(Buffer.from(encoded, "base64"));
284
+ outputs.push({
285
+ data,
286
+ mime: preview.media_type ?? "audio/mpeg",
287
+ ...preview.generated_voice_id === void 0 || preview.generated_voice_id === "" ? {} : { voiceId: preview.generated_voice_id }
288
+ });
289
+ }
290
+ if (outputs.length === 0) throw new AudioGenError("ElevenLabs voice design returned no audio", "audio-empty-result");
291
+ return outputs;
292
+ }
293
+ if (request.mode === "music") {
294
+ const endpoint = `${base}/music`;
295
+ const body = {
296
+ model_id: (request.upstream ?? request.model) || "music_v1",
297
+ prompt: request.prompt,
298
+ ...request.duration !== void 0 && Number.isFinite(request.duration) ? { music_length_ms: Math.round(Math.min(6e5, Math.max(3e3, request.duration * 1e3))) } : {},
299
+ ...request.lyrics !== void 0 && request.lyrics.trim() !== "" ? { lyrics_text: request.lyrics.trim() } : {},
300
+ ...request.isInstrumental !== void 0 ? { force_instrumental: request.isInstrumental } : {}
301
+ };
302
+ const response = await fetchWithTimeout(endpoint, {
303
+ method: "POST",
304
+ redirect: "follow",
305
+ headers,
306
+ body: JSON.stringify(body),
307
+ signal
308
+ }, UPSTREAM_TIMEOUT_MS);
309
+ if (!response.ok) {
310
+ const detail = await response.text().catch(() => "");
311
+ throw new AudioGenError(`ElevenLabs music API error (HTTP ${response.status})${detail === "" ? "" : `: ${detail.slice(0, 300)}`}`, "audio-api-error");
312
+ }
313
+ return normalizeAudioResponse(response, {
314
+ apiKey: channel.apiKey,
315
+ fallbackMime: "audio/mpeg"
316
+ });
317
+ }
318
+ if (request.mode === "sfx") {
319
+ const endpoint = `${base}/sound-generation`;
320
+ const sfxModel = (request.upstream ?? request.model) || "eleven_text_to_sound_v2";
321
+ const body = {
322
+ text: request.prompt,
323
+ model_id: sfxModel,
324
+ ...request.duration !== void 0 && Number.isFinite(request.duration) ? { duration_seconds: Math.min(30, Math.max(.5, request.duration)) } : {},
325
+ ...request.loop !== void 0 ? { loop: request.loop } : {},
326
+ ...request.promptInfluence !== void 0 && Number.isFinite(request.promptInfluence) ? { prompt_influence: Math.min(1, Math.max(0, request.promptInfluence)) } : {}
327
+ };
328
+ const response = await fetchWithTimeout(endpoint, {
329
+ method: "POST",
330
+ redirect: "follow",
331
+ headers,
332
+ body: JSON.stringify(body),
333
+ signal
334
+ }, UPSTREAM_TIMEOUT_MS);
335
+ if (!response.ok) {
336
+ const detail = await response.text().catch(() => "");
337
+ throw new AudioGenError(`ElevenLabs sound effects API error (HTTP ${response.status})${detail === "" ? "" : `: ${detail.slice(0, 300)}`}`, "audio-api-error");
338
+ }
339
+ return normalizeAudioResponse(response, {
340
+ apiKey: channel.apiKey,
341
+ fallbackMime: "audio/mpeg"
342
+ });
343
+ }
253
344
  const voiceId = (request.voice ?? request.model ?? model).trim();
254
345
  const endpoint = `${base}/text-to-speech/${encodeURIComponent(voiceId)}`;
255
346
  const body = {
@@ -266,11 +357,7 @@ async function elevenLabs(channel, request, signal) {
266
357
  return normalizeAudioResponse(await fetchWithTimeout(endpoint, {
267
358
  method: "POST",
268
359
  redirect: "error",
269
- headers: {
270
- "xi-api-key": channel.apiKey.trim(),
271
- "content-type": "application/json",
272
- accept: "audio/mpeg, application/json"
273
- },
360
+ headers,
274
361
  body: JSON.stringify(body),
275
362
  signal
276
363
  }, UPSTREAM_TIMEOUT_MS), {
@@ -563,7 +650,7 @@ async function generateAudio(channel, request, signal) {
563
650
  if (channel.apiUrl.trim() === "") throw new AudioGenError("channel API URL is not configured", "audio-no-endpoint");
564
651
  if (channel.apiKey.trim() === "") throw new AudioGenError("channel API key is not configured", "audio-no-key");
565
652
  if (request.prompt.trim() === "") throw new AudioGenError("audio prompt/text is required", "audio-empty-prompt");
566
- if (request.mode === "voice_design" && !isMiniMax$1(channel)) throw new AudioGenError("音色设计当前仅支持 MiniMax 渠道", "voice-design-unsupported");
653
+ if (request.mode === "voice_design" && !isMiniMax$1(channel) && !isElevenLabs$1(channel)) throw new AudioGenError("音色设计当前仅支持 MiniMax(/v1/voice_design)与 ElevenLabs(/v1/text-to-voice/design)渠道", "voice-design-unsupported");
567
654
  if (isElevenLabs$1(channel)) return elevenLabs(channel, request, signal);
568
655
  if (isMiniMax$1(channel)) return minimax(channel, request, signal);
569
656
  if (isStability$1(channel)) return stabilityAudio(channel, request, signal);
@@ -642,7 +729,7 @@ const AUDIO_PRESETS = [
642
729
  name: "ElevenLabs",
643
730
  apiUrl: "https://api.elevenlabs.io/v1",
644
731
  site: "https://elevenlabsai.cn",
645
- hint: "ElevenLabs 语音合成(TTS);建议点击「获取可用模型」拉取音色与模型",
732
+ hint: "ElevenLabs 语音合成(TTS)与音乐生成(POST /v1/music,music_v2);可点「获取可用模型」拉取音色与模型",
646
733
  models: [
647
734
  {
648
735
  alias: "Rachel",
@@ -678,6 +765,21 @@ const AUDIO_PRESETS = [
678
765
  alias: "eleven_flash_v2_5",
679
766
  id: "eleven_flash_v2_5",
680
767
  category: "tts"
768
+ },
769
+ {
770
+ alias: "music_v2",
771
+ id: "music_v2",
772
+ category: "music"
773
+ },
774
+ {
775
+ alias: "music_v1",
776
+ id: "music_v1",
777
+ category: "music"
778
+ },
779
+ {
780
+ alias: "eleven_text_to_sound_v2",
781
+ id: "eleven_text_to_sound_v2",
782
+ category: "sfx"
681
783
  }
682
784
  ]
683
785
  },
@@ -1078,6 +1180,8 @@ function parseGenerateRequest(body) {
1078
1180
  ...num(body.duration) !== void 0 ? { duration: num(body.duration) } : {},
1079
1181
  ...typeof body.lyrics === "string" && body.lyrics.trim() !== "" ? { lyrics: body.lyrics.trim() } : {},
1080
1182
  ...typeof body.isInstrumental === "boolean" ? { isInstrumental: body.isInstrumental } : {},
1183
+ ...typeof body.loop === "boolean" ? { loop: body.loop } : {},
1184
+ ...num(body.promptInfluence) !== void 0 ? { promptInfluence: num(body.promptInfluence) } : {},
1081
1185
  ...typeof body.format === "string" && body.format.trim() !== "" ? { format: body.format.trim() } : {},
1082
1186
  ...typeof body.channelId === "string" && body.channelId !== "" ? { channelId: body.channelId } : {},
1083
1187
  ...str(body.emotion) !== void 0 ? { emotion: str(body.emotion) } : {},
@@ -1136,22 +1240,24 @@ function resolveChannelRequest(request, view) {
1136
1240
  const defaults = view.channels.find((candidate) => candidate.id === view.defaultChannelId) ?? view.channels[0];
1137
1241
  const target = explicit ?? defaults;
1138
1242
  const asked = request.model.trim();
1243
+ if (request.mode === "voice_design") {
1244
+ if (target === void 0) return {
1245
+ ok: false,
1246
+ code: "no-channels",
1247
+ message: "尚未配置任何渠道"
1248
+ };
1249
+ return {
1250
+ ok: true,
1251
+ request: {
1252
+ ...request,
1253
+ model: "",
1254
+ upstream: void 0,
1255
+ channelId: target.id,
1256
+ channel: target.name
1257
+ }
1258
+ };
1259
+ }
1139
1260
  if (asked === "") {
1140
- if (request.mode === "voice_design") {
1141
- if (target === void 0) return {
1142
- ok: false,
1143
- code: "no-channels",
1144
- message: "尚未配置任何渠道"
1145
- };
1146
- return {
1147
- ok: true,
1148
- request: {
1149
- ...request,
1150
- channelId: target.id,
1151
- channel: target.name
1152
- }
1153
- };
1154
- }
1155
1261
  const alias = target?.models[0]?.alias ?? "";
1156
1262
  if (alias === "") return {
1157
1263
  ok: false,
@@ -1579,7 +1685,7 @@ function ensureConfigured(config) {
1579
1685
  function registerAgentAudioTools(ctx, resolve) {
1580
1686
  return ctx.tools.register(defineTool({
1581
1687
  name: "generate_audio",
1582
- description: "Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and MiniMax voice design. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.",
1688
+ description: "Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and voice design (MiniMax /v1/voice_design, ElevenLabs /v1/text-to-voice/design). The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.",
1583
1689
  parameters: {
1584
1690
  prompt: {
1585
1691
  type: "string",
@@ -1624,6 +1730,14 @@ function registerAgentAudioTools(ctx, resolve) {
1624
1730
  type: "boolean",
1625
1731
  description: "Generate purely instrumental music without vocals/lyrics (MiniMax is_instrumental). When true, lyrics may be omitted."
1626
1732
  },
1733
+ loop: {
1734
+ type: "boolean",
1735
+ description: "Create a seamlessly looping sound effect (ElevenLabs sound generation loop, only for eleven_text_to_sound_v2)."
1736
+ },
1737
+ prompt_influence: {
1738
+ type: "number",
1739
+ description: "Sound effect prompt influence 0-1 (ElevenLabs prompt_influence, default 0.3): higher follows the prompt more closely, lower is more variable."
1740
+ },
1627
1741
  format: {
1628
1742
  type: "string",
1629
1743
  description: "Output format such as mp3 or wav. MiniMax music supports mp3/wav/pcm."
@@ -1764,6 +1878,8 @@ function registerAgentAudioTools(ctx, resolve) {
1764
1878
  ...typeof args.duration === "number" ? { duration: args.duration } : {},
1765
1879
  ...typeof args.lyrics === "string" && args.lyrics.trim() !== "" ? { lyrics: args.lyrics.trim() } : {},
1766
1880
  ...typeof args.is_instrumental === "boolean" ? { isInstrumental: args.is_instrumental } : {},
1881
+ ...typeof args.loop === "boolean" ? { loop: args.loop } : {},
1882
+ ...typeof args.prompt_influence === "number" && Number.isFinite(args.prompt_influence) ? { promptInfluence: args.prompt_influence } : {},
1767
1883
  ...typeof args.format === "string" && args.format.trim() !== "" ? { format: args.format.trim() } : {},
1768
1884
  ...typeof args.emotion === "string" && args.emotion.trim() !== "" ? { emotion: args.emotion.trim() } : {},
1769
1885
  ...typeof args.vol === "number" && Number.isFinite(args.vol) ? { vol: args.vol } : {},
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-audiogen",
3
3
  "description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
4
- "version": "0.3.4",
4
+ "version": "0.3.5",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {
@@ -7,7 +7,13 @@
7
7
  - 帮助用户把音色或音效需求细化成可复用的描述,供后续 TTS/music/sfx 生成。
8
8
  - 输出不建议直接生成音频,而是给出音色参数建议和可用的模型/音色清单。
9
9
 
10
+ ## 面板/工具的音色设计(voice_design)
11
+ - 渠道支持:MiniMax(POST /v1/voice_design,prompt + preview_text)与 ElevenLabs(POST /v1/text-to-voice/design)。
12
+ - MiniMax 参数:`prompt`(音色描述)、`preview_text`(试听文本,任意长度)。
13
+ - ElevenLabs 参数:`voice_description`(音色描述);试听文本 `preview_text` 需 100-1000 字符,过短时自动 `auto_generate_text`;响应含 `previews[].audio_base_64` 与 `generated_voice_id`(可后续在 TTS 中复用该 voice_id)。
14
+ - 面板音色设计模式顶部可切换「厂商 / 渠道」。
15
+
10
16
  ## 流程
11
17
  1. 询问目标风格、音域、情绪、适用场景。
12
18
  2. 生成结构化的音色描述(如温暖复古合成器、未来感 UI 提示音)。
13
- 3. 在后续生成中复用该描述,必要时通过 `generate_audio` 试听。
19
+ 3. 在后续生成中复用该描述,必要时通过 `generate_audio`(mode=voice_design)试听。
@@ -37,3 +37,16 @@
37
37
  ## 常见错误
38
38
  - `lyrics-required`:MiniMax 音乐生成需要歌词,或开启纯音乐。
39
39
  - `HTTP 400` 且含 `2013`:上游参数不合法,检查 lyrics / audio_setting 枚举。
40
+
41
+ ## ElevenLabs Music(POST /v1/music,模型 music_v1 / music_v2)
42
+
43
+ | 字段 | 工具/面板参数 | 说明 |
44
+ | --- | --- | --- |
45
+ | model_id | model(music_v2 等) | 官方枚举:music_v1 / music_v2,默认 music_v1 |
46
+ | prompt | prompt | 音乐/歌词主题描述(不能与 composition_plan 同用;引擎用 prompt) |
47
+ | music_length_ms | duration(秒,自动×1000) | 3000ms - 600000ms(3s-600s),超出自动收敛区间 |
48
+ | lyrics_text | lyrics(歌词) | 选填歌词文本 |
49
+ | force_instrumental | is_instrumental(纯音乐) | true 保证无演唱(无词) |
50
+ | seed / generation_mode / finetune_* | — | 高级字段,暂未透出 |
51
+
52
+ > 响应为音频字节流(audio/*,常为 mp3)。请求同时携带 `xi-api-key` 与 `Authorization: Bearer`,以兼容 New API 类网关(官方站任一头即可)。
@@ -2,15 +2,29 @@
2
2
 
3
3
  ## 触发
4
4
  - `/audio:sfx <描述>`
5
- - 用户说“生成一个音效 / 提示音 / 環境音”
5
+ - 用户说“生成一个音效 / 提示音 / 环境音”
6
6
 
7
7
  ## 参数
8
8
  - prompt: 必填,音效描述
9
- - model: 可选,已配置模型
10
- - duration: 可选,秒数
9
+ - model: 可选,已配置模型(ElevenLabs:eleven_text_to_sound_v2)
10
+ - duration: 可选,秒数(ElevenLabs 0.5-30,留空则自动)
11
+ - loop: 可选,ElevenLabs 无缝循环音效(仅 eleven_text_to_sound_v2)
12
+ - prompt_influence: 可选,ElevenLabs 提示词影响度 0-1(默认 0.3,越高越贴近提示词、越少随机)
11
13
  - format: 可选
12
14
 
13
15
  ## 流程
14
16
  1. 确认已配置音效生成渠道。
15
17
  2. 调用 `generate_audio`,mode=sfx。
16
18
  3. 将音频 URL 返回。
19
+
20
+ ## ElevenLabs Sound Generation(POST /v1/sound-generation)
21
+
22
+ | 字段 | 工具/面板参数 | 说明 |
23
+ | --- | --- | --- |
24
+ | text | prompt | 必填,转换为音效的文本/描述 |
25
+ | model_id | model | 官方枚举:eleven_text_to_sound_v2(默认) |
26
+ | duration_seconds | duration | 0.5-30 秒;留空由提示词推算最优时长 |
27
+ | loop | loop | 是否生成平滑循环音效(仅 eleven_text_to_sound_v2) |
28
+ | prompt_influence | prompt_influence | 0-1,默认 0.3;越高越贴提示词,越低越多样 |
29
+
30
+ > 响应为 audio/mpeg 二进制。请求同时携带 `xi-api-key` 与 `Authorization: Bearer`,兼容 New API 类网关。
@@ -100,7 +100,7 @@ function ensureConfigured(config: AgentAudioToolConfig): void {
100
100
  export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioToolConfig): () => void {
101
101
  const disposer = ctx.tools.register(defineTool({
102
102
  name: 'generate_audio',
103
- description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and MiniMax voice design. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
103
+ description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and voice design (MiniMax /v1/voice_design, ElevenLabs /v1/text-to-voice/design). The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
104
104
  parameters: {
105
105
  prompt: { type: 'string', required: true, description: 'For tts, the text to speak. For music/sfx, a descriptive prompt.' },
106
106
  mode: { type: 'string', enum: ['tts', 'music', 'sfx', 'voice_design'], description: 'Generation mode. Defaults to tts.' },
@@ -111,6 +111,8 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
111
111
  duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
112
112
  lyrics: { type: 'string', description: 'Lyrics for music generation (MiniMax music-3.0/music-cover). Required unless is_instrumental is true. Split verses with an empty line.' },
113
113
  is_instrumental: { type: 'boolean', description: 'Generate purely instrumental music without vocals/lyrics (MiniMax is_instrumental). When true, lyrics may be omitted.' },
114
+ loop: { type: 'boolean', description: 'Create a seamlessly looping sound effect (ElevenLabs sound generation loop, only for eleven_text_to_sound_v2).' },
115
+ prompt_influence: { type: 'number', description: 'Sound effect prompt influence 0-1 (ElevenLabs prompt_influence, default 0.3): higher follows the prompt more closely, lower is more variable.' },
114
116
  format: { type: 'string', description: 'Output format such as mp3 or wav. MiniMax music supports mp3/wav/pcm.' },
115
117
  // ---- MiniMax TTS only (ignored by other providers) ----
116
118
  emotion: { type: 'string', description: 'MiniMax TTS emotion, e.g. happy/sad/angry/nervous/fearful/bored (voice_setting.emotion).' },
@@ -199,6 +201,8 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
199
201
  ...(typeof args.duration === 'number' ? { duration: args.duration } : {}),
200
202
  ...(typeof args.lyrics === 'string' && args.lyrics.trim() !== '' ? { lyrics: args.lyrics.trim() } : {}),
201
203
  ...(typeof args.is_instrumental === 'boolean' ? { isInstrumental: args.is_instrumental } : {}),
204
+ ...(typeof args.loop === 'boolean' ? { loop: args.loop } : {}),
205
+ ...(typeof args.prompt_influence === 'number' && Number.isFinite(args.prompt_influence) ? { promptInfluence: args.prompt_influence } : {}),
202
206
  ...(typeof args.format === 'string' && args.format.trim() !== '' ? { format: args.format.trim() } : {}),
203
207
  // ---- MiniMax TTS 专属字段 ----
204
208
  ...(typeof args.emotion === 'string' && args.emotion.trim() !== '' ? { emotion: args.emotion.trim() } : {}),
@@ -242,6 +242,114 @@ async function openAITTS(channel: AudioChannel, request: GenerateAudioRequest, s
242
242
  async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
243
243
  const base = endpointBase(channel.apiUrl)
244
244
  const model = (request.upstream ?? request.model) || 'eleven_multilingual_v2'
245
+ // 官方使用 xi-api-key;额外携带 Authorization Bearer 以兼容 New API 类网关。
246
+ const headers = {
247
+ 'xi-api-key': channel.apiKey.trim(),
248
+ authorization: `Bearer ${channel.apiKey.trim()}`,
249
+ 'content-type': 'application/json',
250
+ accept: 'audio/mpeg, application/json',
251
+ }
252
+
253
+ // ------------- ElevenLabs Voice Design(POST /v1/text-to-voice/design) -------------
254
+ // voice_description 必填;text 100-1000 字符,过短时用 auto_generate_text;
255
+ // 返回 previews[].audio_base_64 与 previews[].generated_voice_id。
256
+ if (request.mode === 'voice_design') {
257
+ const endpoint = `${base}/text-to-voice/design`
258
+ const previewText = request.previewText?.trim() ?? ''
259
+ const body: Record<string, unknown> = {
260
+ voice_description: request.prompt,
261
+ ...(previewText.length >= 100 ? { text: previewText } : { auto_generate_text: true }),
262
+ }
263
+ const response = await fetchWithTimeout(endpoint, {
264
+ method: 'POST',
265
+ redirect: 'error',
266
+ headers,
267
+ body: JSON.stringify(body),
268
+ signal,
269
+ }, UPSTREAM_TIMEOUT_MS)
270
+ if (!response.ok) {
271
+ const detail = await response.text().catch(() => '')
272
+ throw new AudioGenError(`ElevenLabs voice design API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
273
+ }
274
+ const payload = await response.json() as {
275
+ previews?: Array<{ audio_base_64?: string; generated_voice_id?: string; media_type?: string }>
276
+ }
277
+ const previews = payload.previews ?? []
278
+ if (previews.length === 0) throw new AudioGenError('ElevenLabs voice design returned no previews', 'audio-empty-result')
279
+ const outputs: Array<{ data: Uint8Array; mime: string; voiceId?: string }> = []
280
+ for (const preview of previews) {
281
+ const encoded = preview.audio_base_64?.trim() ?? ''
282
+ if (encoded === '') continue
283
+ const data = new Uint8Array(Buffer.from(encoded, 'base64'))
284
+ outputs.push({
285
+ data,
286
+ mime: preview.media_type ?? 'audio/mpeg',
287
+ ...(preview.generated_voice_id === undefined || preview.generated_voice_id === '' ? {} : { voiceId: preview.generated_voice_id }),
288
+ })
289
+ }
290
+ if (outputs.length === 0) throw new AudioGenError('ElevenLabs voice design returned no audio', 'audio-empty-result')
291
+ return outputs
292
+ }
293
+
294
+ // ------------- ElevenLabs Music(POST /v1/music) -------------
295
+ // 模型:music_v1 / music_v2;prompt 与 composition_plan 二选一(引擎用 prompt)。
296
+ if (request.mode === 'music') {
297
+ const endpoint = `${base}/music`
298
+ const musicModel = (request.upstream ?? request.model) || 'music_v1'
299
+ const body: Record<string, unknown> = {
300
+ model_id: musicModel,
301
+ prompt: request.prompt,
302
+ ...(request.duration !== undefined && Number.isFinite(request.duration)
303
+ ? { music_length_ms: Math.round(Math.min(600_000, Math.max(3_000, request.duration * 1000))) }
304
+ : {}),
305
+ ...(request.lyrics !== undefined && request.lyrics.trim() !== '' ? { lyrics_text: request.lyrics.trim() } : {}),
306
+ ...(request.isInstrumental !== undefined ? { force_instrumental: request.isInstrumental } : {}),
307
+ }
308
+ const response = await fetchWithTimeout(endpoint, {
309
+ method: 'POST',
310
+ redirect: 'follow',
311
+ headers,
312
+ body: JSON.stringify(body),
313
+ signal,
314
+ }, UPSTREAM_TIMEOUT_MS)
315
+ if (!response.ok) {
316
+ const detail = await response.text().catch(() => '')
317
+ throw new AudioGenError(`ElevenLabs music API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
318
+ }
319
+ return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
320
+ }
321
+
322
+ // ------------- ElevenLabs Sound Effects(POST /v1/sound-generation) -------------
323
+ // 官方模型:eleven_text_to_sound_v2;text 必填;duration_seconds 0.5-30;
324
+ // loop 仅该模型可用;prompt_influence 0-1(默认 0.3)。
325
+ if (request.mode === 'sfx') {
326
+ const endpoint = `${base}/sound-generation`
327
+ const sfxModel = (request.upstream ?? request.model) || 'eleven_text_to_sound_v2'
328
+ const body: Record<string, unknown> = {
329
+ text: request.prompt,
330
+ model_id: sfxModel,
331
+ ...(request.duration !== undefined && Number.isFinite(request.duration)
332
+ ? { duration_seconds: Math.min(30, Math.max(0.5, request.duration)) }
333
+ : {}),
334
+ ...(request.loop !== undefined ? { loop: request.loop } : {}),
335
+ ...(request.promptInfluence !== undefined && Number.isFinite(request.promptInfluence)
336
+ ? { prompt_influence: Math.min(1, Math.max(0, request.promptInfluence)) }
337
+ : {}),
338
+ }
339
+ const response = await fetchWithTimeout(endpoint, {
340
+ method: 'POST',
341
+ redirect: 'follow',
342
+ headers,
343
+ body: JSON.stringify(body),
344
+ signal,
345
+ }, UPSTREAM_TIMEOUT_MS)
346
+ if (!response.ok) {
347
+ const detail = await response.text().catch(() => '')
348
+ throw new AudioGenError(`ElevenLabs sound effects API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
349
+ }
350
+ return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
351
+ }
352
+
245
353
  const voiceId = (request.voice ?? request.model ?? model).trim()
246
354
  const endpoint = `${base}/text-to-speech/${encodeURIComponent(voiceId)}`
247
355
  const body: Record<string, unknown> = {
@@ -258,11 +366,7 @@ async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest,
258
366
  const response = await fetchWithTimeout(endpoint, {
259
367
  method: 'POST',
260
368
  redirect: 'error',
261
- headers: {
262
- 'xi-api-key': channel.apiKey.trim(),
263
- 'content-type': 'application/json',
264
- accept: 'audio/mpeg, application/json',
265
- },
369
+ headers,
266
370
  body: JSON.stringify(body),
267
371
  signal,
268
372
  }, UPSTREAM_TIMEOUT_MS)
@@ -591,8 +695,8 @@ export async function generateAudio(
591
695
  if (channel.apiUrl.trim() === '') throw new AudioGenError('channel API URL is not configured', 'audio-no-endpoint')
592
696
  if (channel.apiKey.trim() === '') throw new AudioGenError('channel API key is not configured', 'audio-no-key')
593
697
  if (request.prompt.trim() === '') throw new AudioGenError('audio prompt/text is required', 'audio-empty-prompt')
594
- if (request.mode === 'voice_design' && !isMiniMax(channel)) {
595
- throw new AudioGenError('音色设计当前仅支持 MiniMax 渠道', 'voice-design-unsupported')
698
+ if (request.mode === 'voice_design' && !isMiniMax(channel) && !isElevenLabs(channel)) {
699
+ throw new AudioGenError('音色设计当前仅支持 MiniMax(/v1/voice_design)与 ElevenLabs(/v1/text-to-voice/design)渠道', 'voice-design-unsupported')
596
700
  }
597
701
 
598
702
  if (isElevenLabs(channel)) return elevenLabs(channel, request, signal)
@@ -53,7 +53,7 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
53
53
  name: 'ElevenLabs',
54
54
  apiUrl: 'https://api.elevenlabs.io/v1',
55
55
  site: 'https://elevenlabsai.cn',
56
- hint: 'ElevenLabs 语音合成(TTS);建议点击「获取可用模型」拉取音色与模型',
56
+ hint: 'ElevenLabs 语音合成(TTS)与音乐生成(POST /v1/music,music_v2);可点「获取可用模型」拉取音色与模型',
57
57
  models: [
58
58
  { alias: 'Rachel', id: '21m00Tcm4TlvDq8ikWAM', category: 'tts' },
59
59
  { alias: 'Adam', id: 'pNInz6obpgDQGcFmaJgB', category: 'tts' },
@@ -62,6 +62,11 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
62
62
  { alias: 'eleven_multilingual_v2', id: 'eleven_multilingual_v2', category: 'tts' },
63
63
  { alias: 'eleven_turbo_v2_5', id: 'eleven_turbo_v2_5', category: 'tts' },
64
64
  { alias: 'eleven_flash_v2_5', id: 'eleven_flash_v2_5', category: 'tts' },
65
+ // ElevenLabs Music(POST /v1/music)
66
+ { alias: 'music_v2', id: 'music_v2', category: 'music' },
67
+ { alias: 'music_v1', id: 'music_v1', category: 'music' },
68
+ // ElevenLabs Sound Effects(POST /v1/sound-generation)
69
+ { alias: 'eleven_text_to_sound_v2', id: 'eleven_text_to_sound_v2', category: 'sfx' },
65
70
  ],
66
71
  },
67
72
  {
@@ -62,6 +62,9 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
62
62
  const [duration, setDuration] = useState('')
63
63
  const [lyrics, setLyrics] = useState('')
64
64
  const [instrumental, setInstrumental] = useState(false)
65
+ // ElevenLabs 音效参数
66
+ const [loop, setLoop] = useState(false)
67
+ const [promptInfluence, setPromptInfluence] = useState('')
65
68
  const [format, setFormat] = useState('mp3')
66
69
  // MiniMax TTS 高级参数(其他厂商忽略)
67
70
  const [emotion, setEmotion] = useState('')
@@ -76,12 +79,21 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
76
79
  const [error, setError] = useState<string | null>(null)
77
80
  const [outputs, setOutputs] = useState<GeneratedAudio[]>([])
78
81
  const { entries, reload, clear } = useHistory()
82
+ // 音色设计模式的厂商/渠道选择(默认渠道)
83
+ const [designChannelId, setDesignChannelId] = useState('')
79
84
 
80
85
  const isMiniMaxChannel = useMemo(() => {
81
86
  const target = channels.find(candidate => candidate.id === modelOptions.defaultChannelId) ?? channels[0]
82
87
  return target !== undefined && (target.preset === 'minimax' || /minimax/i.test(target.apiUrl))
83
88
  }, [channels, modelOptions.defaultChannelId])
84
89
 
90
+ useEffect(() => {
91
+ if (channels.length === 0) return
92
+ if (designChannelId === '' || !channels.some(candidate => candidate.id === designChannelId)) {
93
+ setDesignChannelId(modelOptions.defaultChannelId ?? channels[0]!.id)
94
+ }
95
+ }, [channels, modelOptions.defaultChannelId, designChannelId])
96
+
85
97
  const visibleModels = useMemo(() => {
86
98
  if (mode === 'voice_design') return []
87
99
  return modelOptions.models
@@ -105,14 +117,17 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
105
117
  try {
106
118
  const response = await api.generate({
107
119
  mode,
108
- model: (model || visibleModels[0]) ?? '',
120
+ model: mode === 'voice_design' ? '' : (model || visibleModels[0]) ?? '',
109
121
  prompt: prompt.trim(),
122
+ ...(mode === 'voice_design' && designChannelId !== '' ? { channelId: designChannelId } : {}),
110
123
  ...(previewText.trim() !== '' ? { previewText: previewText.trim() } : {}),
111
124
  ...(voice.trim() !== '' ? { voice: voice.trim() } : {}),
112
125
  ...(speed.trim() !== '' ? { speed: Number(speed) } : {}),
113
126
  ...(duration.trim() !== '' ? { duration: Number(duration) } : {}),
114
127
  ...(lyrics.trim() !== '' ? { lyrics: lyrics.trim() } : {}),
115
128
  ...(instrumental ? { isInstrumental: true } : {}),
129
+ ...(loop ? { loop: true } : {}),
130
+ ...(promptInfluence.trim() !== '' ? { promptInfluence: Number(promptInfluence) } : {}),
116
131
  ...(format.trim() !== '' ? { format: format.trim() } : {}),
117
132
  ...(emotion.trim() !== '' ? { emotion: emotion.trim() } : {}),
118
133
  ...(vol.trim() !== '' ? { vol: Number(vol) } : {}),
@@ -173,10 +188,24 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
173
188
  </label>
174
189
 
175
190
  {mode === 'voice_design' ? (
176
- <label className={css.label}>
177
- <span>试听文本</span>
178
- <input className={css.input} value={previewText} onChange={event => setPreviewText(event.target.value)} placeholder="你好,这是新设计的音色试听。" />
179
- </label>
191
+ <>
192
+ <label className={css.label}>
193
+ <span>厂商 / 渠道</span>
194
+ <select className={css.select} value={designChannelId} onChange={event => setDesignChannelId(event.target.value)}>
195
+ {channels.length === 0 ? <option value="">(尚未配置渠道)</option> : null}
196
+ {channels.map(candidate => (
197
+ <option key={candidate.id} value={candidate.id}>
198
+ {candidate.name}({candidate.preset === 'minimax' ? 'MiniMax' : candidate.preset === 'elevenlabs' ? 'ElevenLabs' : candidate.preset || '自定义'})
199
+ </option>
200
+ ))}
201
+ </select>
202
+ </label>
203
+ <p className={css.hint}>MiniMax:/v1/voice_design;ElevenLabs:/v1/text-to-voice/design(试听文本 100-1000 字符,过短将自动生成)</p>
204
+ <label className={css.label}>
205
+ <span>试听文本</span>
206
+ <input className={css.input} value={previewText} onChange={event => setPreviewText(event.target.value)} placeholder="你好,这是新设计的音色试听。" />
207
+ </label>
208
+ </>
180
209
  ) : null}
181
210
 
182
211
  {needModel ? (
@@ -256,6 +285,19 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
256
285
  </label>
257
286
  ) : null}
258
287
 
288
+ {mode === 'sfx' ? (
289
+ <>
290
+ <label className={css.checkbox}>
291
+ <input type="checkbox" checked={loop} onChange={event => setLoop(event.target.checked)} />
292
+ <span>循环音效 loop(无缝循环,需 eleven_text_to_sound_v2)</span>
293
+ </label>
294
+ <label className={css.label}>
295
+ <span>提示词影响度 prompt_influence (0-1)</span>
296
+ <input className={css.input} type="number" step="0.1" min="0" max="1" value={promptInfluence} onChange={event => setPromptInfluence(event.target.value)} placeholder="0.3" />
297
+ </label>
298
+ </>
299
+ ) : null}
300
+
259
301
  {mode === 'music' ? (
260
302
  <>
261
303
  <label className={css.label}>
package/src/protocol.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
9
9
 
10
10
  /** Published package version shared by the host updater and the client UI. */
11
- export const PLUGIN_VERSION = '0.3.4'
11
+ export const PLUGIN_VERSION = '0.3.5'
12
12
 
13
13
  /** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
14
14
  export const SETTINGS_API = {
@@ -118,6 +118,10 @@ export interface GenerateAudioRequest {
118
118
  lyrics?: string
119
119
  /** MiniMax 是否生成纯音乐(无歌词/人声);true 时 lyrics 可为空。 */
120
120
  isInstrumental?: boolean
121
+ /** ElevenLabs 音效:生成无缝循环音效(loop,仅 eleven_text_to_sound_v2)。 */
122
+ loop?: boolean
123
+ /** ElevenLabs 音效:提示词影响度 0-1(prompt_influence,默认 0.3)。 */
124
+ promptInfluence?: number
121
125
  /** Output format, e.g. mp3, wav, pcm. */
122
126
  format?: string
123
127
  /** Channel this request targets (host falls back to default). */
package/src/routes.ts CHANGED
@@ -124,6 +124,8 @@ function parseGenerateRequest(body: Record<string, unknown>): GenerateAudioReque
124
124
  ...(num(body.duration) !== undefined ? { duration: num(body.duration)! } : {}),
125
125
  ...(typeof body.lyrics === 'string' && body.lyrics.trim() !== '' ? { lyrics: body.lyrics.trim() } : {}),
126
126
  ...(typeof body.isInstrumental === 'boolean' ? { isInstrumental: body.isInstrumental } : {}),
127
+ ...(typeof body.loop === 'boolean' ? { loop: body.loop } : {}),
128
+ ...(num(body.promptInfluence) !== undefined ? { promptInfluence: num(body.promptInfluence)! } : {}),
127
129
  ...(typeof body.format === 'string' && body.format.trim() !== '' ? { format: body.format.trim() } : {}),
128
130
  ...(typeof body.channelId === 'string' && body.channelId !== '' ? { channelId: body.channelId } : {}),
129
131
  // ---- MiniMax TTS 专属字段(其他厂商忽略) ----
@@ -180,11 +182,15 @@ function resolveChannelRequest(
180
182
  const defaults = view.channels.find(candidate => candidate.id === view.defaultChannelId) ?? view.channels[0]
181
183
  const target = explicit ?? defaults
182
184
  const asked = request.model.trim()
185
+
186
+ // 音色设计不按模型解析渠道:总是走 channelId(面板选择器)或默认渠道,
187
+ // 并丢弃可能残留的模型值,避免上一个模式的模型把渠道带偏。
188
+ if (request.mode === 'voice_design') {
189
+ if (target === undefined) return { ok: false, code: 'no-channels', message: '尚未配置任何渠道' }
190
+ return { ok: true, request: { ...request, model: '', upstream: undefined, channelId: target.id, channel: target.name } }
191
+ }
192
+
183
193
  if (asked === '') {
184
- if (request.mode === 'voice_design') {
185
- if (target === undefined) return { ok: false, code: 'no-channels', message: '尚未配置任何渠道' }
186
- return { ok: true, request: { ...request, channelId: target.id, channel: target.name } }
187
- }
188
194
  const alias = target?.models[0]?.alias ?? ''
189
195
  if (alias === '') {
190
196
  return { ok: false, code: 'no-models', message: `渠道「${target?.name ?? ''}」尚未配置模型/音色,请先在设置中添加` }