dsh-audiogen 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -22,6 +22,8 @@ const SETTINGS_API = {
22
22
  const GENERATE_API = "/api/dsh-audiogen/generate";
23
23
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
24
24
  const PRESETS_API = "/api/dsh-audiogen/presets";
25
+ /** Host-mediated model/voice discovery endpoint. */
26
+ const MODEL_API = { discover: "/api/dsh-audiogen/models/discover" };
25
27
  /** Loopback-only audio file reader for panel/tool-result previews. */
26
28
  const AUDIO_API = { file: "/api/dsh-audiogen/audio" };
27
29
  /** Host-persisted generation history routes. */
@@ -91,7 +93,7 @@ function isOpenAICompatible(channel, mode) {
91
93
  function isElevenLabs(channel) {
92
94
  return isPreset(channel, "elevenlabs") || /elevenlabs/i.test(channel.apiUrl);
93
95
  }
94
- function isMiniMax(channel) {
96
+ function isMiniMax$1(channel) {
95
97
  return isPreset(channel, "minimax") || /minimax/i.test(channel.apiUrl);
96
98
  }
97
99
  function isStability(channel) {
@@ -183,13 +185,14 @@ async function normalizeAudioResponse(response, options) {
183
185
  } catch {
184
186
  throw new AudioGenError("audio endpoint returned an unprocessable response body", "audio-bad-response");
185
187
  }
186
- const base64 = findBase64Audio(parsed);
187
- if (base64 !== void 0 && base64.length > 0) {
188
+ const encoded = findBase64Audio(parsed);
189
+ if (encoded !== void 0 && encoded.length > 0) {
188
190
  let data;
189
191
  try {
190
- data = new Uint8Array(Buffer.from(base64, "base64"));
192
+ const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0;
193
+ data = new Uint8Array(Buffer.from(encoded, isHex ? "hex" : "base64"));
191
194
  } catch {
192
- throw new AudioGenError("audio endpoint returned invalid base64", "audio-bad-response");
195
+ throw new AudioGenError("audio endpoint returned invalid audio encoding", "audio-bad-response");
193
196
  }
194
197
  return [{
195
198
  data,
@@ -274,27 +277,79 @@ async function elevenLabs(channel, request, signal) {
274
277
  fallbackMime: "audio/mpeg"
275
278
  });
276
279
  }
280
+ function minimaxApiBase(base) {
281
+ const trimmed = endpointBase(base);
282
+ return /\/v1$/i.test(trimmed) ? trimmed : `${trimmed}/v1`;
283
+ }
277
284
  async function minimax(channel, request, signal) {
278
- const base = endpointBase(channel.apiUrl);
279
- const endpoint = /\/t2a_v2(\?|$)/i.test(base) ? base : `${base}/t2a_v2`;
280
- const model = (request.upstream ?? request.model) || "speech-01-turbo";
285
+ const base = minimaxApiBase(channel.apiUrl);
286
+ const model = (request.upstream ?? request.model) || (request.mode === "music" ? "music-3.0" : "speech-2.8-hd");
281
287
  const voice = request.voice ?? request.model ?? "";
282
- const body = {
283
- model,
284
- text: request.prompt,
285
- stream: false,
286
- ...voice === "" ? {} : { voice_setting: {
287
- voice_id: voice,
288
- ...request.speed !== void 0 ? { speed: request.speed } : {},
289
- vol: 1,
290
- pitch: 0
291
- } },
292
- audio_setting: {
293
- format: request.format ?? "mp3",
294
- sample_rate: 32e3,
295
- bitrate: 128e3
288
+ if (request.mode === "voice_design") {
289
+ const endpoint = `${base}/voice_design`;
290
+ const body = {
291
+ prompt: request.prompt,
292
+ preview_text: request.previewText ?? request.voice ?? "你好,这是新设计的音色试听。"
293
+ };
294
+ const response = await fetchWithTimeout(endpoint, {
295
+ method: "POST",
296
+ redirect: "error",
297
+ headers: {
298
+ authorization: `Bearer ${channel.apiKey.trim()}`,
299
+ "content-type": "application/json",
300
+ accept: "application/json"
301
+ },
302
+ body: JSON.stringify(body),
303
+ signal
304
+ }, UPSTREAM_TIMEOUT_MS);
305
+ if (!response.ok) {
306
+ const detail = await response.text().catch(() => "");
307
+ throw new AudioGenError(`MiniMax voice design API error (HTTP ${response.status})${detail === "" ? "" : `: ${detail.slice(0, 300)}`}`, "audio-api-error");
296
308
  }
297
- };
309
+ const payload = await response.json();
310
+ if (payload.base_resp?.status_code !== void 0 && payload.base_resp.status_code !== 0) throw new AudioGenError(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`, "audio-api-error");
311
+ const encoded = payload.trial_audio ?? "";
312
+ if (encoded === "") throw new AudioGenError("MiniMax voice design returned no trial audio", "audio-empty-result");
313
+ const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0;
314
+ return [{
315
+ data: new Uint8Array(Buffer.from(encoded, isHex ? "hex" : "base64")),
316
+ mime: "audio/mpeg",
317
+ ...payload.voice_id === void 0 ? {} : { voiceId: payload.voice_id }
318
+ }];
319
+ }
320
+ let endpoint;
321
+ let body;
322
+ if (request.mode === "music") {
323
+ endpoint = `${base}/music_generation`;
324
+ body = {
325
+ model,
326
+ prompt: request.prompt,
327
+ ...request.duration !== void 0 ? { duration: request.duration } : {},
328
+ audio_setting: {
329
+ format: request.format ?? "mp3",
330
+ sample_rate: 44100,
331
+ bitrate: 256e3
332
+ }
333
+ };
334
+ } else {
335
+ endpoint = `${base}/t2a_v2`;
336
+ body = {
337
+ model,
338
+ text: request.prompt,
339
+ stream: false,
340
+ ...voice === "" ? {} : { voice_setting: {
341
+ voice_id: voice,
342
+ ...request.speed !== void 0 ? { speed: request.speed } : {},
343
+ vol: 1,
344
+ pitch: 0
345
+ } },
346
+ audio_setting: {
347
+ format: request.format ?? "mp3",
348
+ sample_rate: 32e3,
349
+ bitrate: 128e3
350
+ }
351
+ };
352
+ }
298
353
  return normalizeAudioResponse(await fetchWithTimeout(endpoint, {
299
354
  method: "POST",
300
355
  redirect: "error",
@@ -369,8 +424,9 @@ async function generateAudio(channel, request, signal) {
369
424
  if (channel.apiUrl.trim() === "") throw new AudioGenError("channel API URL is not configured", "audio-no-endpoint");
370
425
  if (channel.apiKey.trim() === "") throw new AudioGenError("channel API key is not configured", "audio-no-key");
371
426
  if (request.prompt.trim() === "") throw new AudioGenError("audio prompt/text is required", "audio-empty-prompt");
427
+ if (request.mode === "voice_design" && !isMiniMax$1(channel)) throw new AudioGenError("音色设计当前仅支持 MiniMax 渠道", "voice-design-unsupported");
372
428
  if (isElevenLabs(channel)) return elevenLabs(channel, request, signal);
373
- if (isMiniMax(channel)) return minimax(channel, request, signal);
429
+ if (isMiniMax$1(channel)) return minimax(channel, request, signal);
374
430
  if (isStability(channel)) return stabilityAudio(channel, request, signal);
375
431
  if (isOpenAICompatible(channel, request.mode)) return openAITTS(channel, request, signal);
376
432
  return genericAudio(channel, request, signal);
@@ -386,15 +442,18 @@ const AUDIO_PRESETS = [
386
442
  models: [
387
443
  {
388
444
  alias: "tts-1",
389
- id: "tts-1"
445
+ id: "tts-1",
446
+ category: "tts"
390
447
  },
391
448
  {
392
449
  alias: "tts-1-hd",
393
- id: "tts-1-hd"
450
+ id: "tts-1-hd",
451
+ category: "tts"
394
452
  },
395
453
  {
396
454
  alias: "gpt-4o-mini-tts",
397
- id: "gpt-4o-mini-tts"
455
+ id: "gpt-4o-mini-tts",
456
+ category: "tts"
398
457
  }
399
458
  ]
400
459
  },
@@ -406,43 +465,86 @@ const AUDIO_PRESETS = [
406
465
  models: [
407
466
  {
408
467
  alias: "Rachel",
409
- id: "21m00Tcm4TlvDq8ikWAM"
468
+ id: "21m00Tcm4TlvDq8ikWAM",
469
+ category: "tts"
410
470
  },
411
471
  {
412
472
  alias: "Adam",
413
- id: "pNInz6obpgDQGcFmaJgB"
473
+ id: "pNInz6obpgDQGcFmaJgB",
474
+ category: "tts"
414
475
  },
415
476
  {
416
477
  alias: "Antoni",
417
- id: "ErXwobaYiN019PkySvjV"
478
+ id: "ErXwobaYiN019PkySvjV",
479
+ category: "tts"
418
480
  },
419
481
  {
420
482
  alias: "Bella",
421
- id: "EXAVITQu4vr4xnSDxMaL"
483
+ id: "EXAVITQu4vr4xnSDxMaL",
484
+ category: "tts"
422
485
  }
423
486
  ]
424
487
  },
425
488
  {
426
489
  id: "minimax",
427
490
  name: "MiniMax",
428
- apiUrl: "https://api.minimax.chat/v1",
429
- hint: "MiniMax 语音合成(T2A);需在 API URL 后按官方要求携带 GroupId 或使用完整接口地址",
491
+ apiUrl: "https://api.minimaxi.com",
492
+ hint: "MiniMax 音色设计 / TTS / 音乐生成;可使用“获取可用模型”拉取账号音色",
430
493
  models: [
431
494
  {
432
- alias: "speech-01-turbo",
433
- id: "speech-01-turbo"
495
+ alias: "speech-2.8-hd",
496
+ id: "speech-2.8-hd",
497
+ category: "tts"
434
498
  },
435
499
  {
436
- alias: "speech-01-hd",
437
- id: "speech-01-hd"
500
+ alias: "speech-2.8-turbo",
501
+ id: "speech-2.8-turbo",
502
+ category: "tts"
438
503
  },
439
504
  {
440
- alias: "speech-02-turbo",
441
- id: "speech-02-turbo"
505
+ alias: "speech-2.6-hd",
506
+ id: "speech-2.6-hd",
507
+ category: "tts"
508
+ },
509
+ {
510
+ alias: "speech-2.6-turbo",
511
+ id: "speech-2.6-turbo",
512
+ category: "tts"
442
513
  },
443
514
  {
444
515
  alias: "speech-02-hd",
445
- id: "speech-02-hd"
516
+ id: "speech-02-hd",
517
+ category: "tts"
518
+ },
519
+ {
520
+ alias: "speech-02-turbo",
521
+ id: "speech-02-turbo",
522
+ category: "tts"
523
+ },
524
+ {
525
+ alias: "speech-01-hd",
526
+ id: "speech-01-hd",
527
+ category: "tts"
528
+ },
529
+ {
530
+ alias: "speech-01-turbo",
531
+ id: "speech-01-turbo",
532
+ category: "tts"
533
+ },
534
+ {
535
+ alias: "music-3.0",
536
+ id: "music-3.0",
537
+ category: "music"
538
+ },
539
+ {
540
+ alias: "music-2.6",
541
+ id: "music-2.6",
542
+ category: "music"
543
+ },
544
+ {
545
+ alias: "music-cover",
546
+ id: "music-cover",
547
+ category: "music"
446
548
  }
447
549
  ]
448
550
  },
@@ -453,10 +555,12 @@ const AUDIO_PRESETS = [
453
555
  hint: "Stability AI 音乐/音效生成(stable-audio 系列)",
454
556
  models: [{
455
557
  alias: "stable-audio-2.0",
456
- id: "stable-audio-2.0"
558
+ id: "stable-audio-2.0",
559
+ category: "music"
457
560
  }, {
458
561
  alias: "stable-audio-1.0",
459
- id: "stable-audio-1.0"
562
+ id: "stable-audio-1.0",
563
+ category: "music"
460
564
  }]
461
565
  },
462
566
  {
@@ -472,6 +576,113 @@ function audioPresetById(id) {
472
576
  return AUDIO_PRESETS.find((preset) => preset.id === id);
473
577
  }
474
578
  //#endregion
579
+ //#region src/audio-models.ts
580
+ function isMiniMax(channel) {
581
+ return channel.preset === "minimax" || /minimax/i.test(channel.apiUrl);
582
+ }
583
+ function baseUrl(url) {
584
+ return url.trim().replace(/\/+$/, "");
585
+ }
586
+ function categoryFor(id) {
587
+ const value = id.toLowerCase();
588
+ if (/(tts|speech|voice|t2a)/i.test(value)) return "tts";
589
+ if (/(music|song|cover|lyrics)/i.test(value)) return "music";
590
+ if (/(sfx|sound.?effect|effect|foley)/i.test(value)) return "sfx";
591
+ }
592
+ async function postJson(url, apiKey, body) {
593
+ const response = await fetch(url, {
594
+ method: "POST",
595
+ headers: {
596
+ authorization: `Bearer ${apiKey.trim()}`,
597
+ "content-type": "application/json"
598
+ },
599
+ body: JSON.stringify(body)
600
+ });
601
+ if (!response.ok) {
602
+ const text = await response.text().catch(() => "");
603
+ throw new Error(`HTTP ${response.status}${text === "" ? "" : `: ${text.slice(0, 300)}`}`);
604
+ }
605
+ return response.json();
606
+ }
607
+ /** Discover available models/voices for a channel. */
608
+ async function discoverAudioModels(channel) {
609
+ if (channel.apiUrl.trim() === "") throw new Error("API URL is not configured");
610
+ if (channel.apiKey.trim() === "") throw new Error("API key is not configured");
611
+ if (isMiniMax(channel)) {
612
+ const payload = await postJson(`${baseUrl(channel.apiUrl).replace(/\/v1$/i, "")}/v1/get_voice`, channel.apiKey, { voice_type: "all" });
613
+ if (payload.base_resp?.status_code !== void 0 && payload.base_resp.status_code !== 0) throw new Error(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`);
614
+ const models = [];
615
+ for (const voice of payload.system_voice ?? []) {
616
+ const id = voice.voice_id?.trim() ?? "";
617
+ if (id === "") continue;
618
+ models.push({
619
+ alias: voice.voice_name?.trim() || id,
620
+ id,
621
+ category: "tts",
622
+ ...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
623
+ });
624
+ }
625
+ for (const voice of payload.voice_cloning ?? []) {
626
+ const id = voice.voice_id?.trim() ?? "";
627
+ if (id === "") continue;
628
+ models.push({
629
+ alias: id,
630
+ id,
631
+ category: "tts",
632
+ ...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
633
+ });
634
+ }
635
+ for (const voice of payload.voice_generation ?? []) {
636
+ const id = voice.voice_id?.trim() ?? "";
637
+ if (id === "") continue;
638
+ models.push({
639
+ alias: id,
640
+ id,
641
+ category: "tts",
642
+ ...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
643
+ });
644
+ }
645
+ const music = (audioPresetById("minimax")?.models ?? []).filter((model) => model.category === "music");
646
+ for (const model of music) models.push({
647
+ ...model,
648
+ category: "music"
649
+ });
650
+ return {
651
+ models: dedupe(models),
652
+ source: "MiniMax get_voice + built-in music catalog"
653
+ };
654
+ }
655
+ const url = `${baseUrl(channel.apiUrl)}/models`;
656
+ const response = await fetch(url, { headers: { authorization: `Bearer ${channel.apiKey.trim()}` } });
657
+ if (!response.ok) throw new Error(`model list request failed (HTTP ${response.status}); please add models manually`);
658
+ const payload = await response.json();
659
+ const models = [];
660
+ for (const item of payload.data ?? []) {
661
+ const id = item.id?.trim() ?? "";
662
+ if (id === "") continue;
663
+ const category = categoryFor(id) ?? "tts";
664
+ models.push({
665
+ alias: id,
666
+ id,
667
+ category
668
+ });
669
+ }
670
+ return {
671
+ models: dedupe(models),
672
+ source: "OpenAI-compatible /models"
673
+ };
674
+ }
675
+ function dedupe(models) {
676
+ const seen = /* @__PURE__ */ new Set();
677
+ const out = [];
678
+ for (const model of models) {
679
+ if (seen.has(model.id)) continue;
680
+ seen.add(model.id);
681
+ out.push(model);
682
+ }
683
+ return out;
684
+ }
685
+ //#endregion
475
686
  //#region src/audio-store.ts
476
687
  /**
477
688
  * Host-side persistence for generated audio and generation history.
@@ -553,13 +764,15 @@ async function appendHistory(entry) {
553
764
  model: entry.model,
554
765
  prompt: entry.prompt,
555
766
  ...entry.voice === void 0 ? {} : { voice: entry.voice },
767
+ ...entry.voiceId === void 0 ? {} : { voiceId: entry.voiceId },
556
768
  ...entry.speed === void 0 ? {} : { speed: entry.speed },
557
769
  ...entry.duration === void 0 ? {} : { duration: entry.duration },
558
770
  ...entry.format === void 0 ? {} : { format: entry.format },
559
771
  audio: entry.audio.map((audio) => ({
560
772
  url: audio.url,
561
773
  mime: audio.mime,
562
- ...audio.duration === void 0 ? {} : { duration: audio.duration }
774
+ ...audio.duration === void 0 ? {} : { duration: audio.duration },
775
+ ...audio.voiceId === void 0 ? {} : { voiceId: audio.voiceId }
563
776
  })),
564
777
  ...entry.channelId === void 0 ? {} : { channelId: entry.channelId },
565
778
  ...entry.channel === void 0 ? {} : { channel: entry.channel }
@@ -631,7 +844,7 @@ function messageOf(error) {
631
844
  return error instanceof Error ? error.message : String(error);
632
845
  }
633
846
  function parseGenerateRequest(body) {
634
- const mode = body.mode === "music" ? "music" : body.mode === "sfx" ? "sfx" : "tts";
847
+ const mode = body.mode === "music" ? "music" : body.mode === "sfx" ? "sfx" : body.mode === "voice_design" ? "voice_design" : "tts";
635
848
  const prompt = typeof body.prompt === "string" ? body.prompt.trim() : "";
636
849
  if (prompt === "") return void 0;
637
850
  return {
@@ -639,6 +852,7 @@ function parseGenerateRequest(body) {
639
852
  model: typeof body.model === "string" ? body.model.trim() : "",
640
853
  prompt,
641
854
  ...typeof body.voice === "string" && body.voice.trim() !== "" ? { voice: body.voice.trim() } : {},
855
+ ...typeof body.previewText === "string" && body.previewText.trim() !== "" ? { previewText: body.previewText.trim() } : {},
642
856
  ...typeof body.speed === "number" ? { speed: body.speed } : {},
643
857
  ...typeof body.duration === "number" ? { duration: body.duration } : {},
644
858
  ...typeof body.format === "string" && body.format.trim() !== "" ? { format: body.format.trim() } : {},
@@ -685,6 +899,21 @@ function resolveChannelRequest(request, view) {
685
899
  const target = explicit ?? defaults;
686
900
  const asked = request.model.trim();
687
901
  if (asked === "") {
902
+ if (request.mode === "voice_design") {
903
+ if (target === void 0) return {
904
+ ok: false,
905
+ code: "no-channels",
906
+ message: "尚未配置任何渠道"
907
+ };
908
+ return {
909
+ ok: true,
910
+ request: {
911
+ ...request,
912
+ channelId: target.id,
913
+ channel: target.name
914
+ }
915
+ };
916
+ }
688
917
  const alias = target?.models[0]?.alias ?? "";
689
918
  if (alias === "") return {
690
919
  ok: false,
@@ -758,6 +987,36 @@ function makeRoutes(deps) {
758
987
  });
759
988
  }
760
989
  },
990
+ {
991
+ kind: "exact",
992
+ path: MODEL_API.discover,
993
+ handler: async (req, res) => {
994
+ if (!guard(req, res, "POST")) return;
995
+ const body = await readJsonBody(req);
996
+ const view = deps.resolveChannels();
997
+ const stored = view.channels.find((candidate) => candidate.id === (typeof body?.channelId === "string" ? body.channelId : void 0)) ?? view.channels.find((candidate) => candidate.id === view.defaultChannelId) ?? view.channels[0];
998
+ const channel = {
999
+ id: stored?.id ?? "preview",
1000
+ preset: stored?.preset ?? "",
1001
+ name: stored?.name ?? "",
1002
+ apiUrl: typeof body?.apiUrl === "string" && body.apiUrl.trim() !== "" ? body.apiUrl.trim() : stored?.apiUrl ?? "",
1003
+ apiKey: typeof body?.apiKey === "string" && body.apiKey.trim() !== "" ? body.apiKey.trim() : stored?.apiKey ?? "",
1004
+ models: stored?.models ?? []
1005
+ };
1006
+ try {
1007
+ writeJson(res, 200, {
1008
+ ok: true,
1009
+ ...await discoverAudioModels(channel)
1010
+ });
1011
+ } catch (error) {
1012
+ writeJson(res, 200, {
1013
+ ok: false,
1014
+ code: "model-discovery-failed",
1015
+ message: messageOf(error)
1016
+ });
1017
+ }
1018
+ }
1019
+ },
761
1020
  {
762
1021
  kind: "exact",
763
1022
  path: SETTINGS_API.describe,
@@ -855,7 +1114,8 @@ function makeRoutes(deps) {
855
1114
  b64: Buffer.from(output.data).toString("base64"),
856
1115
  mime: saved.mime,
857
1116
  bytes: saved.bytes,
858
- url: `${AUDIO_API.file}/${encodeURIComponent(saved.file)}`
1117
+ url: `${AUDIO_API.file}/${encodeURIComponent(saved.file)}`,
1118
+ ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
859
1119
  });
860
1120
  }
861
1121
  let history;
@@ -1012,7 +1272,8 @@ const resultSchema = {
1012
1272
  enum: [
1013
1273
  "tts",
1014
1274
  "music",
1015
- "sfx"
1275
+ "sfx",
1276
+ "voice_design"
1016
1277
  ]
1017
1278
  },
1018
1279
  model: {
@@ -1041,7 +1302,8 @@ const resultSchema = {
1041
1302
  bytes: {
1042
1303
  type: "integer",
1043
1304
  required: true
1044
- }
1305
+ },
1306
+ voiceId: { type: "string" }
1045
1307
  }
1046
1308
  }
1047
1309
  },
@@ -1079,7 +1341,7 @@ function ensureConfigured(config) {
1079
1341
  function registerAgentAudioTools(ctx, resolve) {
1080
1342
  return ctx.tools.register(defineTool({
1081
1343
  name: "generate_audio",
1082
- description: "Generate audio with the configured audio provider. Supports text-to-speech, music generation and sound effects. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.",
1344
+ description: "Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and MiniMax voice design. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.",
1083
1345
  parameters: {
1084
1346
  prompt: {
1085
1347
  type: "string",
@@ -1091,7 +1353,8 @@ function registerAgentAudioTools(ctx, resolve) {
1091
1353
  enum: [
1092
1354
  "tts",
1093
1355
  "music",
1094
- "sfx"
1356
+ "sfx",
1357
+ "voice_design"
1095
1358
  ],
1096
1359
  description: "Generation mode. Defaults to tts."
1097
1360
  },
@@ -1103,6 +1366,10 @@ function registerAgentAudioTools(ctx, resolve) {
1103
1366
  type: "string",
1104
1367
  description: "Optional voice id/name for TTS providers."
1105
1368
  },
1369
+ preview_text: {
1370
+ type: "string",
1371
+ description: "Optional preview text for voice_design."
1372
+ },
1106
1373
  speed: {
1107
1374
  type: "number",
1108
1375
  description: "Optional speaking rate / speed multiplier where supported."
@@ -1125,15 +1392,26 @@ function registerAgentAudioTools(ctx, resolve) {
1125
1392
  async execute(args, exec) {
1126
1393
  const config = resolve();
1127
1394
  ensureConfigured(config);
1128
- const picked = resolveModel(config, args.model);
1395
+ const mode = args.mode === "music" ? "music" : args.mode === "sfx" ? "sfx" : args.mode === "voice_design" ? "voice_design" : "tts";
1396
+ const picked = mode === "voice_design" ? (() => {
1397
+ const usable = config.channels.filter((channel) => channel.apiUrl.trim() !== "" && channel.apiKey.trim() !== "");
1398
+ const target = usable.find((channel) => channel.id === config.defaultChannelId) ?? usable[0];
1399
+ if (target === void 0) throw new AudioGenError("No usable audio channel is configured for voice design.", "no-channel-available");
1400
+ return {
1401
+ channel: target,
1402
+ alias: "",
1403
+ upstream: ""
1404
+ };
1405
+ })() : resolveModel(config, args.model);
1129
1406
  const request = {
1130
- mode: args.mode === "music" ? "music" : args.mode === "sfx" ? "sfx" : "tts",
1407
+ mode,
1131
1408
  model: picked.alias,
1132
1409
  upstream: picked.upstream,
1133
1410
  channelId: picked.channel.id,
1134
1411
  channel: picked.channel.name,
1135
1412
  prompt: args.prompt.trim(),
1136
1413
  ...typeof args.voice === "string" && args.voice.trim() !== "" ? { voice: args.voice.trim() } : {},
1414
+ ...typeof args.preview_text === "string" && args.preview_text.trim() !== "" ? { previewText: args.preview_text.trim() } : {},
1137
1415
  ...typeof args.speed === "number" ? { speed: args.speed } : {},
1138
1416
  ...typeof args.duration === "number" ? { duration: args.duration } : {},
1139
1417
  ...typeof args.format === "string" && args.format.trim() !== "" ? { format: args.format.trim() } : {}
@@ -1147,7 +1425,8 @@ function registerAgentAudioTools(ctx, resolve) {
1147
1425
  id: saved.id,
1148
1426
  url: `/api/dsh-audiogen/audio/${encodeURIComponent(saved.file)}`,
1149
1427
  mime: saved.mime,
1150
- bytes: saved.bytes
1428
+ bytes: saved.bytes,
1429
+ ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
1151
1430
  });
1152
1431
  }
1153
1432
  try {
@@ -1166,7 +1445,8 @@ function registerAgentAudioTools(ctx, resolve) {
1166
1445
  b64: Buffer.from(output.data).toString("base64"),
1167
1446
  mime: audio[index].mime,
1168
1447
  bytes: audio[index].bytes,
1169
- url: audio[index].url
1448
+ url: audio[index].url,
1449
+ ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
1170
1450
  })),
1171
1451
  channelId: picked.channel.id,
1172
1452
  channel: picked.channel.name
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-audiogen",
3
3
  "description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
4
- "version": "0.1.0",
4
+ "version": "0.3.0",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {
@@ -24,6 +24,7 @@ interface AgentAudioRef {
24
24
  url: string
25
25
  mime: string
26
26
  bytes: number
27
+ voiceId?: string
27
28
  }
28
29
 
29
30
  interface AgentAudioResult {
@@ -43,6 +44,7 @@ const audioRefSchema = {
43
44
  url: { type: 'string', required: true },
44
45
  mime: { type: 'string', required: true },
45
46
  bytes: { type: 'integer', required: true },
47
+ voiceId: { type: 'string' },
46
48
  },
47
49
  } as const
48
50
 
@@ -52,7 +54,7 @@ const resultSchema = {
52
54
  properties: {
53
55
  status: { type: 'string', required: true },
54
56
  message: { type: 'string', required: true },
55
- mode: { type: 'string', required: true, enum: ['tts', 'music', 'sfx'] },
57
+ mode: { type: 'string', required: true, enum: ['tts', 'music', 'sfx', 'voice_design'] },
56
58
  model: { type: 'string', required: true },
57
59
  audio: { type: 'array', required: true, items: audioRefSchema },
58
60
  error: { type: 'string' },
@@ -98,12 +100,13 @@ function ensureConfigured(config: AgentAudioToolConfig): void {
98
100
  export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioToolConfig): () => void {
99
101
  const disposer = ctx.tools.register(defineTool({
100
102
  name: 'generate_audio',
101
- description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation and sound effects. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
103
+ description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and MiniMax voice design. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
102
104
  parameters: {
103
105
  prompt: { type: 'string', required: true, description: 'For tts, the text to speak. For music/sfx, a descriptive prompt.' },
104
- mode: { type: 'string', enum: ['tts', 'music', 'sfx'], description: 'Generation mode. Defaults to tts.' },
106
+ mode: { type: 'string', enum: ['tts', 'music', 'sfx', 'voice_design'], description: 'Generation mode. Defaults to tts.' },
105
107
  model: { type: 'string', description: 'One of the configured audio models/voices. Defaults to the first configured model.' },
106
108
  voice: { type: 'string', description: 'Optional voice id/name for TTS providers.' },
109
+ preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
107
110
  speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported.' },
108
111
  duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
109
112
  format: { type: 'string', description: 'Output format such as mp3 or wav.' },
@@ -117,15 +120,24 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
117
120
  async execute(args, exec) {
118
121
  const config = resolve()
119
122
  ensureConfigured(config)
120
- const picked = resolveModel(config, args.model)
123
+ const mode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
124
+ const picked = mode === 'voice_design'
125
+ ? (() => {
126
+ const usable = config.channels.filter(channel => channel.apiUrl.trim() !== '' && channel.apiKey.trim() !== '')
127
+ const target = usable.find(channel => channel.id === config.defaultChannelId) ?? usable[0]
128
+ if (target === undefined) throw new AudioGenError('No usable audio channel is configured for voice design.', 'no-channel-available')
129
+ return { channel: target, alias: '', upstream: '' }
130
+ })()
131
+ : resolveModel(config, args.model)
121
132
  const request: GenerateAudioRequest = {
122
- mode: args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : 'tts',
133
+ mode,
123
134
  model: picked.alias,
124
135
  upstream: picked.upstream,
125
136
  channelId: picked.channel.id,
126
137
  channel: picked.channel.name,
127
138
  prompt: args.prompt.trim(),
128
139
  ...(typeof args.voice === 'string' && args.voice.trim() !== '' ? { voice: args.voice.trim() } : {}),
140
+ ...(typeof args.preview_text === 'string' && args.preview_text.trim() !== '' ? { previewText: args.preview_text.trim() } : {}),
129
141
  ...(typeof args.speed === 'number' ? { speed: args.speed } : {}),
130
142
  ...(typeof args.duration === 'number' ? { duration: args.duration } : {}),
131
143
  ...(typeof args.format === 'string' && args.format.trim() !== '' ? { format: args.format.trim() } : {}),
@@ -140,6 +152,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
140
152
  url: `/api/dsh-audiogen/audio/${encodeURIComponent(saved.file)}`,
141
153
  mime: saved.mime,
142
154
  bytes: saved.bytes,
155
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
143
156
  })
144
157
  }
145
158
  try {
@@ -159,6 +172,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
159
172
  mime: audio[index]!.mime,
160
173
  bytes: audio[index]!.bytes,
161
174
  url: audio[index]!.url,
175
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
162
176
  })),
163
177
  channelId: picked.channel.id,
164
178
  channel: picked.channel.name,