dsh-audiogen 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/lib/client.js +373 -291
- package/lib/client.js.map +1 -1
- package/lib/index.js +333 -53
- package/package.json +1 -1
- package/src/agent-audio-tools.ts +19 -5
- package/src/audio-engine.ts +96 -29
- package/src/audio-models.ts +129 -0
- package/src/audio-presets.ts +24 -15
- package/src/audio-store.ts +2 -0
- package/src/client/AudioGenPanel.tsx +76 -36
- package/src/client/SettingsCard.tsx +52 -5
- package/src/client/channels-form.ts +5 -1
- package/src/client/locales.ts +6 -4
- package/src/client/settings-scope.ts +10 -4
- package/src/protocol.ts +31 -2
- package/src/routes.ts +35 -2
package/lib/index.js
CHANGED
|
@@ -22,6 +22,8 @@ const SETTINGS_API = {
|
|
|
22
22
|
const GENERATE_API = "/api/dsh-audiogen/generate";
|
|
23
23
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
24
24
|
const PRESETS_API = "/api/dsh-audiogen/presets";
|
|
25
|
+
/** Host-mediated model/voice discovery endpoint. */
|
|
26
|
+
const MODEL_API = { discover: "/api/dsh-audiogen/models/discover" };
|
|
25
27
|
/** Loopback-only audio file reader for panel/tool-result previews. */
|
|
26
28
|
const AUDIO_API = { file: "/api/dsh-audiogen/audio" };
|
|
27
29
|
/** Host-persisted generation history routes. */
|
|
@@ -91,7 +93,7 @@ function isOpenAICompatible(channel, mode) {
|
|
|
91
93
|
function isElevenLabs(channel) {
|
|
92
94
|
return isPreset(channel, "elevenlabs") || /elevenlabs/i.test(channel.apiUrl);
|
|
93
95
|
}
|
|
94
|
-
function isMiniMax(channel) {
|
|
96
|
+
function isMiniMax$1(channel) {
|
|
95
97
|
return isPreset(channel, "minimax") || /minimax/i.test(channel.apiUrl);
|
|
96
98
|
}
|
|
97
99
|
function isStability(channel) {
|
|
@@ -183,13 +185,14 @@ async function normalizeAudioResponse(response, options) {
|
|
|
183
185
|
} catch {
|
|
184
186
|
throw new AudioGenError("audio endpoint returned an unprocessable response body", "audio-bad-response");
|
|
185
187
|
}
|
|
186
|
-
const
|
|
187
|
-
if (
|
|
188
|
+
const encoded = findBase64Audio(parsed);
|
|
189
|
+
if (encoded !== void 0 && encoded.length > 0) {
|
|
188
190
|
let data;
|
|
189
191
|
try {
|
|
190
|
-
|
|
192
|
+
const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0;
|
|
193
|
+
data = new Uint8Array(Buffer.from(encoded, isHex ? "hex" : "base64"));
|
|
191
194
|
} catch {
|
|
192
|
-
throw new AudioGenError("audio endpoint returned invalid
|
|
195
|
+
throw new AudioGenError("audio endpoint returned invalid audio encoding", "audio-bad-response");
|
|
193
196
|
}
|
|
194
197
|
return [{
|
|
195
198
|
data,
|
|
@@ -274,27 +277,79 @@ async function elevenLabs(channel, request, signal) {
|
|
|
274
277
|
fallbackMime: "audio/mpeg"
|
|
275
278
|
});
|
|
276
279
|
}
|
|
280
|
+
function minimaxApiBase(base) {
|
|
281
|
+
const trimmed = endpointBase(base);
|
|
282
|
+
return /\/v1$/i.test(trimmed) ? trimmed : `${trimmed}/v1`;
|
|
283
|
+
}
|
|
277
284
|
async function minimax(channel, request, signal) {
|
|
278
|
-
const base =
|
|
279
|
-
const
|
|
280
|
-
const model = (request.upstream ?? request.model) || "speech-01-turbo";
|
|
285
|
+
const base = minimaxApiBase(channel.apiUrl);
|
|
286
|
+
const model = (request.upstream ?? request.model) || (request.mode === "music" ? "music-3.0" : "speech-2.8-hd");
|
|
281
287
|
const voice = request.voice ?? request.model ?? "";
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
288
|
+
if (request.mode === "voice_design") {
|
|
289
|
+
const endpoint = `${base}/voice_design`;
|
|
290
|
+
const body = {
|
|
291
|
+
prompt: request.prompt,
|
|
292
|
+
preview_text: request.previewText ?? request.voice ?? "你好,这是新设计的音色试听。"
|
|
293
|
+
};
|
|
294
|
+
const response = await fetchWithTimeout(endpoint, {
|
|
295
|
+
method: "POST",
|
|
296
|
+
redirect: "error",
|
|
297
|
+
headers: {
|
|
298
|
+
authorization: `Bearer ${channel.apiKey.trim()}`,
|
|
299
|
+
"content-type": "application/json",
|
|
300
|
+
accept: "application/json"
|
|
301
|
+
},
|
|
302
|
+
body: JSON.stringify(body),
|
|
303
|
+
signal
|
|
304
|
+
}, UPSTREAM_TIMEOUT_MS);
|
|
305
|
+
if (!response.ok) {
|
|
306
|
+
const detail = await response.text().catch(() => "");
|
|
307
|
+
throw new AudioGenError(`MiniMax voice design API error (HTTP ${response.status})${detail === "" ? "" : `: ${detail.slice(0, 300)}`}`, "audio-api-error");
|
|
296
308
|
}
|
|
297
|
-
|
|
309
|
+
const payload = await response.json();
|
|
310
|
+
if (payload.base_resp?.status_code !== void 0 && payload.base_resp.status_code !== 0) throw new AudioGenError(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`, "audio-api-error");
|
|
311
|
+
const encoded = payload.trial_audio ?? "";
|
|
312
|
+
if (encoded === "") throw new AudioGenError("MiniMax voice design returned no trial audio", "audio-empty-result");
|
|
313
|
+
const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0;
|
|
314
|
+
return [{
|
|
315
|
+
data: new Uint8Array(Buffer.from(encoded, isHex ? "hex" : "base64")),
|
|
316
|
+
mime: "audio/mpeg",
|
|
317
|
+
...payload.voice_id === void 0 ? {} : { voiceId: payload.voice_id }
|
|
318
|
+
}];
|
|
319
|
+
}
|
|
320
|
+
let endpoint;
|
|
321
|
+
let body;
|
|
322
|
+
if (request.mode === "music") {
|
|
323
|
+
endpoint = `${base}/music_generation`;
|
|
324
|
+
body = {
|
|
325
|
+
model,
|
|
326
|
+
prompt: request.prompt,
|
|
327
|
+
...request.duration !== void 0 ? { duration: request.duration } : {},
|
|
328
|
+
audio_setting: {
|
|
329
|
+
format: request.format ?? "mp3",
|
|
330
|
+
sample_rate: 44100,
|
|
331
|
+
bitrate: 256e3
|
|
332
|
+
}
|
|
333
|
+
};
|
|
334
|
+
} else {
|
|
335
|
+
endpoint = `${base}/t2a_v2`;
|
|
336
|
+
body = {
|
|
337
|
+
model,
|
|
338
|
+
text: request.prompt,
|
|
339
|
+
stream: false,
|
|
340
|
+
...voice === "" ? {} : { voice_setting: {
|
|
341
|
+
voice_id: voice,
|
|
342
|
+
...request.speed !== void 0 ? { speed: request.speed } : {},
|
|
343
|
+
vol: 1,
|
|
344
|
+
pitch: 0
|
|
345
|
+
} },
|
|
346
|
+
audio_setting: {
|
|
347
|
+
format: request.format ?? "mp3",
|
|
348
|
+
sample_rate: 32e3,
|
|
349
|
+
bitrate: 128e3
|
|
350
|
+
}
|
|
351
|
+
};
|
|
352
|
+
}
|
|
298
353
|
return normalizeAudioResponse(await fetchWithTimeout(endpoint, {
|
|
299
354
|
method: "POST",
|
|
300
355
|
redirect: "error",
|
|
@@ -369,8 +424,9 @@ async function generateAudio(channel, request, signal) {
|
|
|
369
424
|
if (channel.apiUrl.trim() === "") throw new AudioGenError("channel API URL is not configured", "audio-no-endpoint");
|
|
370
425
|
if (channel.apiKey.trim() === "") throw new AudioGenError("channel API key is not configured", "audio-no-key");
|
|
371
426
|
if (request.prompt.trim() === "") throw new AudioGenError("audio prompt/text is required", "audio-empty-prompt");
|
|
427
|
+
if (request.mode === "voice_design" && !isMiniMax$1(channel)) throw new AudioGenError("音色设计当前仅支持 MiniMax 渠道", "voice-design-unsupported");
|
|
372
428
|
if (isElevenLabs(channel)) return elevenLabs(channel, request, signal);
|
|
373
|
-
if (isMiniMax(channel)) return minimax(channel, request, signal);
|
|
429
|
+
if (isMiniMax$1(channel)) return minimax(channel, request, signal);
|
|
374
430
|
if (isStability(channel)) return stabilityAudio(channel, request, signal);
|
|
375
431
|
if (isOpenAICompatible(channel, request.mode)) return openAITTS(channel, request, signal);
|
|
376
432
|
return genericAudio(channel, request, signal);
|
|
@@ -386,15 +442,18 @@ const AUDIO_PRESETS = [
|
|
|
386
442
|
models: [
|
|
387
443
|
{
|
|
388
444
|
alias: "tts-1",
|
|
389
|
-
id: "tts-1"
|
|
445
|
+
id: "tts-1",
|
|
446
|
+
category: "tts"
|
|
390
447
|
},
|
|
391
448
|
{
|
|
392
449
|
alias: "tts-1-hd",
|
|
393
|
-
id: "tts-1-hd"
|
|
450
|
+
id: "tts-1-hd",
|
|
451
|
+
category: "tts"
|
|
394
452
|
},
|
|
395
453
|
{
|
|
396
454
|
alias: "gpt-4o-mini-tts",
|
|
397
|
-
id: "gpt-4o-mini-tts"
|
|
455
|
+
id: "gpt-4o-mini-tts",
|
|
456
|
+
category: "tts"
|
|
398
457
|
}
|
|
399
458
|
]
|
|
400
459
|
},
|
|
@@ -406,43 +465,86 @@ const AUDIO_PRESETS = [
|
|
|
406
465
|
models: [
|
|
407
466
|
{
|
|
408
467
|
alias: "Rachel",
|
|
409
|
-
id: "21m00Tcm4TlvDq8ikWAM"
|
|
468
|
+
id: "21m00Tcm4TlvDq8ikWAM",
|
|
469
|
+
category: "tts"
|
|
410
470
|
},
|
|
411
471
|
{
|
|
412
472
|
alias: "Adam",
|
|
413
|
-
id: "pNInz6obpgDQGcFmaJgB"
|
|
473
|
+
id: "pNInz6obpgDQGcFmaJgB",
|
|
474
|
+
category: "tts"
|
|
414
475
|
},
|
|
415
476
|
{
|
|
416
477
|
alias: "Antoni",
|
|
417
|
-
id: "ErXwobaYiN019PkySvjV"
|
|
478
|
+
id: "ErXwobaYiN019PkySvjV",
|
|
479
|
+
category: "tts"
|
|
418
480
|
},
|
|
419
481
|
{
|
|
420
482
|
alias: "Bella",
|
|
421
|
-
id: "EXAVITQu4vr4xnSDxMaL"
|
|
483
|
+
id: "EXAVITQu4vr4xnSDxMaL",
|
|
484
|
+
category: "tts"
|
|
422
485
|
}
|
|
423
486
|
]
|
|
424
487
|
},
|
|
425
488
|
{
|
|
426
489
|
id: "minimax",
|
|
427
490
|
name: "MiniMax",
|
|
428
|
-
apiUrl: "https://api.
|
|
429
|
-
hint: "MiniMax
|
|
491
|
+
apiUrl: "https://api.minimaxi.com",
|
|
492
|
+
hint: "MiniMax 音色设计 / TTS / 音乐生成;可使用“获取可用模型”拉取账号音色",
|
|
430
493
|
models: [
|
|
431
494
|
{
|
|
432
|
-
alias: "speech-
|
|
433
|
-
id: "speech-
|
|
495
|
+
alias: "speech-2.8-hd",
|
|
496
|
+
id: "speech-2.8-hd",
|
|
497
|
+
category: "tts"
|
|
434
498
|
},
|
|
435
499
|
{
|
|
436
|
-
alias: "speech-
|
|
437
|
-
id: "speech-
|
|
500
|
+
alias: "speech-2.8-turbo",
|
|
501
|
+
id: "speech-2.8-turbo",
|
|
502
|
+
category: "tts"
|
|
438
503
|
},
|
|
439
504
|
{
|
|
440
|
-
alias: "speech-
|
|
441
|
-
id: "speech-
|
|
505
|
+
alias: "speech-2.6-hd",
|
|
506
|
+
id: "speech-2.6-hd",
|
|
507
|
+
category: "tts"
|
|
508
|
+
},
|
|
509
|
+
{
|
|
510
|
+
alias: "speech-2.6-turbo",
|
|
511
|
+
id: "speech-2.6-turbo",
|
|
512
|
+
category: "tts"
|
|
442
513
|
},
|
|
443
514
|
{
|
|
444
515
|
alias: "speech-02-hd",
|
|
445
|
-
id: "speech-02-hd"
|
|
516
|
+
id: "speech-02-hd",
|
|
517
|
+
category: "tts"
|
|
518
|
+
},
|
|
519
|
+
{
|
|
520
|
+
alias: "speech-02-turbo",
|
|
521
|
+
id: "speech-02-turbo",
|
|
522
|
+
category: "tts"
|
|
523
|
+
},
|
|
524
|
+
{
|
|
525
|
+
alias: "speech-01-hd",
|
|
526
|
+
id: "speech-01-hd",
|
|
527
|
+
category: "tts"
|
|
528
|
+
},
|
|
529
|
+
{
|
|
530
|
+
alias: "speech-01-turbo",
|
|
531
|
+
id: "speech-01-turbo",
|
|
532
|
+
category: "tts"
|
|
533
|
+
},
|
|
534
|
+
{
|
|
535
|
+
alias: "music-3.0",
|
|
536
|
+
id: "music-3.0",
|
|
537
|
+
category: "music"
|
|
538
|
+
},
|
|
539
|
+
{
|
|
540
|
+
alias: "music-2.6",
|
|
541
|
+
id: "music-2.6",
|
|
542
|
+
category: "music"
|
|
543
|
+
},
|
|
544
|
+
{
|
|
545
|
+
alias: "music-cover",
|
|
546
|
+
id: "music-cover",
|
|
547
|
+
category: "music"
|
|
446
548
|
}
|
|
447
549
|
]
|
|
448
550
|
},
|
|
@@ -453,10 +555,12 @@ const AUDIO_PRESETS = [
|
|
|
453
555
|
hint: "Stability AI 音乐/音效生成(stable-audio 系列)",
|
|
454
556
|
models: [{
|
|
455
557
|
alias: "stable-audio-2.0",
|
|
456
|
-
id: "stable-audio-2.0"
|
|
558
|
+
id: "stable-audio-2.0",
|
|
559
|
+
category: "music"
|
|
457
560
|
}, {
|
|
458
561
|
alias: "stable-audio-1.0",
|
|
459
|
-
id: "stable-audio-1.0"
|
|
562
|
+
id: "stable-audio-1.0",
|
|
563
|
+
category: "music"
|
|
460
564
|
}]
|
|
461
565
|
},
|
|
462
566
|
{
|
|
@@ -472,6 +576,113 @@ function audioPresetById(id) {
|
|
|
472
576
|
return AUDIO_PRESETS.find((preset) => preset.id === id);
|
|
473
577
|
}
|
|
474
578
|
//#endregion
|
|
579
|
+
//#region src/audio-models.ts
|
|
580
|
+
function isMiniMax(channel) {
|
|
581
|
+
return channel.preset === "minimax" || /minimax/i.test(channel.apiUrl);
|
|
582
|
+
}
|
|
583
|
+
function baseUrl(url) {
|
|
584
|
+
return url.trim().replace(/\/+$/, "");
|
|
585
|
+
}
|
|
586
|
+
function categoryFor(id) {
|
|
587
|
+
const value = id.toLowerCase();
|
|
588
|
+
if (/(tts|speech|voice|t2a)/i.test(value)) return "tts";
|
|
589
|
+
if (/(music|song|cover|lyrics)/i.test(value)) return "music";
|
|
590
|
+
if (/(sfx|sound.?effect|effect|foley)/i.test(value)) return "sfx";
|
|
591
|
+
}
|
|
592
|
+
async function postJson(url, apiKey, body) {
|
|
593
|
+
const response = await fetch(url, {
|
|
594
|
+
method: "POST",
|
|
595
|
+
headers: {
|
|
596
|
+
authorization: `Bearer ${apiKey.trim()}`,
|
|
597
|
+
"content-type": "application/json"
|
|
598
|
+
},
|
|
599
|
+
body: JSON.stringify(body)
|
|
600
|
+
});
|
|
601
|
+
if (!response.ok) {
|
|
602
|
+
const text = await response.text().catch(() => "");
|
|
603
|
+
throw new Error(`HTTP ${response.status}${text === "" ? "" : `: ${text.slice(0, 300)}`}`);
|
|
604
|
+
}
|
|
605
|
+
return response.json();
|
|
606
|
+
}
|
|
607
|
+
/** Discover available models/voices for a channel. */
|
|
608
|
+
async function discoverAudioModels(channel) {
|
|
609
|
+
if (channel.apiUrl.trim() === "") throw new Error("API URL is not configured");
|
|
610
|
+
if (channel.apiKey.trim() === "") throw new Error("API key is not configured");
|
|
611
|
+
if (isMiniMax(channel)) {
|
|
612
|
+
const payload = await postJson(`${baseUrl(channel.apiUrl).replace(/\/v1$/i, "")}/v1/get_voice`, channel.apiKey, { voice_type: "all" });
|
|
613
|
+
if (payload.base_resp?.status_code !== void 0 && payload.base_resp.status_code !== 0) throw new Error(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`);
|
|
614
|
+
const models = [];
|
|
615
|
+
for (const voice of payload.system_voice ?? []) {
|
|
616
|
+
const id = voice.voice_id?.trim() ?? "";
|
|
617
|
+
if (id === "") continue;
|
|
618
|
+
models.push({
|
|
619
|
+
alias: voice.voice_name?.trim() || id,
|
|
620
|
+
id,
|
|
621
|
+
category: "tts",
|
|
622
|
+
...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
|
|
623
|
+
});
|
|
624
|
+
}
|
|
625
|
+
for (const voice of payload.voice_cloning ?? []) {
|
|
626
|
+
const id = voice.voice_id?.trim() ?? "";
|
|
627
|
+
if (id === "") continue;
|
|
628
|
+
models.push({
|
|
629
|
+
alias: id,
|
|
630
|
+
id,
|
|
631
|
+
category: "tts",
|
|
632
|
+
...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
|
|
633
|
+
});
|
|
634
|
+
}
|
|
635
|
+
for (const voice of payload.voice_generation ?? []) {
|
|
636
|
+
const id = voice.voice_id?.trim() ?? "";
|
|
637
|
+
if (id === "") continue;
|
|
638
|
+
models.push({
|
|
639
|
+
alias: id,
|
|
640
|
+
id,
|
|
641
|
+
category: "tts",
|
|
642
|
+
...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
|
|
643
|
+
});
|
|
644
|
+
}
|
|
645
|
+
const music = (audioPresetById("minimax")?.models ?? []).filter((model) => model.category === "music");
|
|
646
|
+
for (const model of music) models.push({
|
|
647
|
+
...model,
|
|
648
|
+
category: "music"
|
|
649
|
+
});
|
|
650
|
+
return {
|
|
651
|
+
models: dedupe(models),
|
|
652
|
+
source: "MiniMax get_voice + built-in music catalog"
|
|
653
|
+
};
|
|
654
|
+
}
|
|
655
|
+
const url = `${baseUrl(channel.apiUrl)}/models`;
|
|
656
|
+
const response = await fetch(url, { headers: { authorization: `Bearer ${channel.apiKey.trim()}` } });
|
|
657
|
+
if (!response.ok) throw new Error(`model list request failed (HTTP ${response.status}); please add models manually`);
|
|
658
|
+
const payload = await response.json();
|
|
659
|
+
const models = [];
|
|
660
|
+
for (const item of payload.data ?? []) {
|
|
661
|
+
const id = item.id?.trim() ?? "";
|
|
662
|
+
if (id === "") continue;
|
|
663
|
+
const category = categoryFor(id) ?? "tts";
|
|
664
|
+
models.push({
|
|
665
|
+
alias: id,
|
|
666
|
+
id,
|
|
667
|
+
category
|
|
668
|
+
});
|
|
669
|
+
}
|
|
670
|
+
return {
|
|
671
|
+
models: dedupe(models),
|
|
672
|
+
source: "OpenAI-compatible /models"
|
|
673
|
+
};
|
|
674
|
+
}
|
|
675
|
+
function dedupe(models) {
|
|
676
|
+
const seen = /* @__PURE__ */ new Set();
|
|
677
|
+
const out = [];
|
|
678
|
+
for (const model of models) {
|
|
679
|
+
if (seen.has(model.id)) continue;
|
|
680
|
+
seen.add(model.id);
|
|
681
|
+
out.push(model);
|
|
682
|
+
}
|
|
683
|
+
return out;
|
|
684
|
+
}
|
|
685
|
+
//#endregion
|
|
475
686
|
//#region src/audio-store.ts
|
|
476
687
|
/**
|
|
477
688
|
* Host-side persistence for generated audio and generation history.
|
|
@@ -553,13 +764,15 @@ async function appendHistory(entry) {
|
|
|
553
764
|
model: entry.model,
|
|
554
765
|
prompt: entry.prompt,
|
|
555
766
|
...entry.voice === void 0 ? {} : { voice: entry.voice },
|
|
767
|
+
...entry.voiceId === void 0 ? {} : { voiceId: entry.voiceId },
|
|
556
768
|
...entry.speed === void 0 ? {} : { speed: entry.speed },
|
|
557
769
|
...entry.duration === void 0 ? {} : { duration: entry.duration },
|
|
558
770
|
...entry.format === void 0 ? {} : { format: entry.format },
|
|
559
771
|
audio: entry.audio.map((audio) => ({
|
|
560
772
|
url: audio.url,
|
|
561
773
|
mime: audio.mime,
|
|
562
|
-
...audio.duration === void 0 ? {} : { duration: audio.duration }
|
|
774
|
+
...audio.duration === void 0 ? {} : { duration: audio.duration },
|
|
775
|
+
...audio.voiceId === void 0 ? {} : { voiceId: audio.voiceId }
|
|
563
776
|
})),
|
|
564
777
|
...entry.channelId === void 0 ? {} : { channelId: entry.channelId },
|
|
565
778
|
...entry.channel === void 0 ? {} : { channel: entry.channel }
|
|
@@ -631,7 +844,7 @@ function messageOf(error) {
|
|
|
631
844
|
return error instanceof Error ? error.message : String(error);
|
|
632
845
|
}
|
|
633
846
|
function parseGenerateRequest(body) {
|
|
634
|
-
const mode = body.mode === "music" ? "music" : body.mode === "sfx" ? "sfx" : "tts";
|
|
847
|
+
const mode = body.mode === "music" ? "music" : body.mode === "sfx" ? "sfx" : body.mode === "voice_design" ? "voice_design" : "tts";
|
|
635
848
|
const prompt = typeof body.prompt === "string" ? body.prompt.trim() : "";
|
|
636
849
|
if (prompt === "") return void 0;
|
|
637
850
|
return {
|
|
@@ -639,6 +852,7 @@ function parseGenerateRequest(body) {
|
|
|
639
852
|
model: typeof body.model === "string" ? body.model.trim() : "",
|
|
640
853
|
prompt,
|
|
641
854
|
...typeof body.voice === "string" && body.voice.trim() !== "" ? { voice: body.voice.trim() } : {},
|
|
855
|
+
...typeof body.previewText === "string" && body.previewText.trim() !== "" ? { previewText: body.previewText.trim() } : {},
|
|
642
856
|
...typeof body.speed === "number" ? { speed: body.speed } : {},
|
|
643
857
|
...typeof body.duration === "number" ? { duration: body.duration } : {},
|
|
644
858
|
...typeof body.format === "string" && body.format.trim() !== "" ? { format: body.format.trim() } : {},
|
|
@@ -685,6 +899,21 @@ function resolveChannelRequest(request, view) {
|
|
|
685
899
|
const target = explicit ?? defaults;
|
|
686
900
|
const asked = request.model.trim();
|
|
687
901
|
if (asked === "") {
|
|
902
|
+
if (request.mode === "voice_design") {
|
|
903
|
+
if (target === void 0) return {
|
|
904
|
+
ok: false,
|
|
905
|
+
code: "no-channels",
|
|
906
|
+
message: "尚未配置任何渠道"
|
|
907
|
+
};
|
|
908
|
+
return {
|
|
909
|
+
ok: true,
|
|
910
|
+
request: {
|
|
911
|
+
...request,
|
|
912
|
+
channelId: target.id,
|
|
913
|
+
channel: target.name
|
|
914
|
+
}
|
|
915
|
+
};
|
|
916
|
+
}
|
|
688
917
|
const alias = target?.models[0]?.alias ?? "";
|
|
689
918
|
if (alias === "") return {
|
|
690
919
|
ok: false,
|
|
@@ -758,6 +987,36 @@ function makeRoutes(deps) {
|
|
|
758
987
|
});
|
|
759
988
|
}
|
|
760
989
|
},
|
|
990
|
+
{
|
|
991
|
+
kind: "exact",
|
|
992
|
+
path: MODEL_API.discover,
|
|
993
|
+
handler: async (req, res) => {
|
|
994
|
+
if (!guard(req, res, "POST")) return;
|
|
995
|
+
const body = await readJsonBody(req);
|
|
996
|
+
const view = deps.resolveChannels();
|
|
997
|
+
const stored = view.channels.find((candidate) => candidate.id === (typeof body?.channelId === "string" ? body.channelId : void 0)) ?? view.channels.find((candidate) => candidate.id === view.defaultChannelId) ?? view.channels[0];
|
|
998
|
+
const channel = {
|
|
999
|
+
id: stored?.id ?? "preview",
|
|
1000
|
+
preset: stored?.preset ?? "",
|
|
1001
|
+
name: stored?.name ?? "",
|
|
1002
|
+
apiUrl: typeof body?.apiUrl === "string" && body.apiUrl.trim() !== "" ? body.apiUrl.trim() : stored?.apiUrl ?? "",
|
|
1003
|
+
apiKey: typeof body?.apiKey === "string" && body.apiKey.trim() !== "" ? body.apiKey.trim() : stored?.apiKey ?? "",
|
|
1004
|
+
models: stored?.models ?? []
|
|
1005
|
+
};
|
|
1006
|
+
try {
|
|
1007
|
+
writeJson(res, 200, {
|
|
1008
|
+
ok: true,
|
|
1009
|
+
...await discoverAudioModels(channel)
|
|
1010
|
+
});
|
|
1011
|
+
} catch (error) {
|
|
1012
|
+
writeJson(res, 200, {
|
|
1013
|
+
ok: false,
|
|
1014
|
+
code: "model-discovery-failed",
|
|
1015
|
+
message: messageOf(error)
|
|
1016
|
+
});
|
|
1017
|
+
}
|
|
1018
|
+
}
|
|
1019
|
+
},
|
|
761
1020
|
{
|
|
762
1021
|
kind: "exact",
|
|
763
1022
|
path: SETTINGS_API.describe,
|
|
@@ -855,7 +1114,8 @@ function makeRoutes(deps) {
|
|
|
855
1114
|
b64: Buffer.from(output.data).toString("base64"),
|
|
856
1115
|
mime: saved.mime,
|
|
857
1116
|
bytes: saved.bytes,
|
|
858
|
-
url: `${AUDIO_API.file}/${encodeURIComponent(saved.file)}
|
|
1117
|
+
url: `${AUDIO_API.file}/${encodeURIComponent(saved.file)}`,
|
|
1118
|
+
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
859
1119
|
});
|
|
860
1120
|
}
|
|
861
1121
|
let history;
|
|
@@ -1012,7 +1272,8 @@ const resultSchema = {
|
|
|
1012
1272
|
enum: [
|
|
1013
1273
|
"tts",
|
|
1014
1274
|
"music",
|
|
1015
|
-
"sfx"
|
|
1275
|
+
"sfx",
|
|
1276
|
+
"voice_design"
|
|
1016
1277
|
]
|
|
1017
1278
|
},
|
|
1018
1279
|
model: {
|
|
@@ -1041,7 +1302,8 @@ const resultSchema = {
|
|
|
1041
1302
|
bytes: {
|
|
1042
1303
|
type: "integer",
|
|
1043
1304
|
required: true
|
|
1044
|
-
}
|
|
1305
|
+
},
|
|
1306
|
+
voiceId: { type: "string" }
|
|
1045
1307
|
}
|
|
1046
1308
|
}
|
|
1047
1309
|
},
|
|
@@ -1079,7 +1341,7 @@ function ensureConfigured(config) {
|
|
|
1079
1341
|
function registerAgentAudioTools(ctx, resolve) {
|
|
1080
1342
|
return ctx.tools.register(defineTool({
|
|
1081
1343
|
name: "generate_audio",
|
|
1082
|
-
description: "Generate audio with the configured audio provider. Supports text-to-speech, music generation and
|
|
1344
|
+
description: "Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and MiniMax voice design. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.",
|
|
1083
1345
|
parameters: {
|
|
1084
1346
|
prompt: {
|
|
1085
1347
|
type: "string",
|
|
@@ -1091,7 +1353,8 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
1091
1353
|
enum: [
|
|
1092
1354
|
"tts",
|
|
1093
1355
|
"music",
|
|
1094
|
-
"sfx"
|
|
1356
|
+
"sfx",
|
|
1357
|
+
"voice_design"
|
|
1095
1358
|
],
|
|
1096
1359
|
description: "Generation mode. Defaults to tts."
|
|
1097
1360
|
},
|
|
@@ -1103,6 +1366,10 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
1103
1366
|
type: "string",
|
|
1104
1367
|
description: "Optional voice id/name for TTS providers."
|
|
1105
1368
|
},
|
|
1369
|
+
preview_text: {
|
|
1370
|
+
type: "string",
|
|
1371
|
+
description: "Optional preview text for voice_design."
|
|
1372
|
+
},
|
|
1106
1373
|
speed: {
|
|
1107
1374
|
type: "number",
|
|
1108
1375
|
description: "Optional speaking rate / speed multiplier where supported."
|
|
@@ -1125,15 +1392,26 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
1125
1392
|
async execute(args, exec) {
|
|
1126
1393
|
const config = resolve();
|
|
1127
1394
|
ensureConfigured(config);
|
|
1128
|
-
const
|
|
1395
|
+
const mode = args.mode === "music" ? "music" : args.mode === "sfx" ? "sfx" : args.mode === "voice_design" ? "voice_design" : "tts";
|
|
1396
|
+
const picked = mode === "voice_design" ? (() => {
|
|
1397
|
+
const usable = config.channels.filter((channel) => channel.apiUrl.trim() !== "" && channel.apiKey.trim() !== "");
|
|
1398
|
+
const target = usable.find((channel) => channel.id === config.defaultChannelId) ?? usable[0];
|
|
1399
|
+
if (target === void 0) throw new AudioGenError("No usable audio channel is configured for voice design.", "no-channel-available");
|
|
1400
|
+
return {
|
|
1401
|
+
channel: target,
|
|
1402
|
+
alias: "",
|
|
1403
|
+
upstream: ""
|
|
1404
|
+
};
|
|
1405
|
+
})() : resolveModel(config, args.model);
|
|
1129
1406
|
const request = {
|
|
1130
|
-
mode
|
|
1407
|
+
mode,
|
|
1131
1408
|
model: picked.alias,
|
|
1132
1409
|
upstream: picked.upstream,
|
|
1133
1410
|
channelId: picked.channel.id,
|
|
1134
1411
|
channel: picked.channel.name,
|
|
1135
1412
|
prompt: args.prompt.trim(),
|
|
1136
1413
|
...typeof args.voice === "string" && args.voice.trim() !== "" ? { voice: args.voice.trim() } : {},
|
|
1414
|
+
...typeof args.preview_text === "string" && args.preview_text.trim() !== "" ? { previewText: args.preview_text.trim() } : {},
|
|
1137
1415
|
...typeof args.speed === "number" ? { speed: args.speed } : {},
|
|
1138
1416
|
...typeof args.duration === "number" ? { duration: args.duration } : {},
|
|
1139
1417
|
...typeof args.format === "string" && args.format.trim() !== "" ? { format: args.format.trim() } : {}
|
|
@@ -1147,7 +1425,8 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
1147
1425
|
id: saved.id,
|
|
1148
1426
|
url: `/api/dsh-audiogen/audio/${encodeURIComponent(saved.file)}`,
|
|
1149
1427
|
mime: saved.mime,
|
|
1150
|
-
bytes: saved.bytes
|
|
1428
|
+
bytes: saved.bytes,
|
|
1429
|
+
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
1151
1430
|
});
|
|
1152
1431
|
}
|
|
1153
1432
|
try {
|
|
@@ -1166,7 +1445,8 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
1166
1445
|
b64: Buffer.from(output.data).toString("base64"),
|
|
1167
1446
|
mime: audio[index].mime,
|
|
1168
1447
|
bytes: audio[index].bytes,
|
|
1169
|
-
url: audio[index].url
|
|
1448
|
+
url: audio[index].url,
|
|
1449
|
+
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
1170
1450
|
})),
|
|
1171
1451
|
channelId: picked.channel.id,
|
|
1172
1452
|
channel: picked.channel.name
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-audiogen",
|
|
3
3
|
"description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.3.0",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|
package/src/agent-audio-tools.ts
CHANGED
|
@@ -24,6 +24,7 @@ interface AgentAudioRef {
|
|
|
24
24
|
url: string
|
|
25
25
|
mime: string
|
|
26
26
|
bytes: number
|
|
27
|
+
voiceId?: string
|
|
27
28
|
}
|
|
28
29
|
|
|
29
30
|
interface AgentAudioResult {
|
|
@@ -43,6 +44,7 @@ const audioRefSchema = {
|
|
|
43
44
|
url: { type: 'string', required: true },
|
|
44
45
|
mime: { type: 'string', required: true },
|
|
45
46
|
bytes: { type: 'integer', required: true },
|
|
47
|
+
voiceId: { type: 'string' },
|
|
46
48
|
},
|
|
47
49
|
} as const
|
|
48
50
|
|
|
@@ -52,7 +54,7 @@ const resultSchema = {
|
|
|
52
54
|
properties: {
|
|
53
55
|
status: { type: 'string', required: true },
|
|
54
56
|
message: { type: 'string', required: true },
|
|
55
|
-
mode: { type: 'string', required: true, enum: ['tts', 'music', 'sfx'] },
|
|
57
|
+
mode: { type: 'string', required: true, enum: ['tts', 'music', 'sfx', 'voice_design'] },
|
|
56
58
|
model: { type: 'string', required: true },
|
|
57
59
|
audio: { type: 'array', required: true, items: audioRefSchema },
|
|
58
60
|
error: { type: 'string' },
|
|
@@ -98,12 +100,13 @@ function ensureConfigured(config: AgentAudioToolConfig): void {
|
|
|
98
100
|
export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioToolConfig): () => void {
|
|
99
101
|
const disposer = ctx.tools.register(defineTool({
|
|
100
102
|
name: 'generate_audio',
|
|
101
|
-
description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation and
|
|
103
|
+
description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and MiniMax voice design. The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
|
|
102
104
|
parameters: {
|
|
103
105
|
prompt: { type: 'string', required: true, description: 'For tts, the text to speak. For music/sfx, a descriptive prompt.' },
|
|
104
|
-
mode: { type: 'string', enum: ['tts', 'music', 'sfx'], description: 'Generation mode. Defaults to tts.' },
|
|
106
|
+
mode: { type: 'string', enum: ['tts', 'music', 'sfx', 'voice_design'], description: 'Generation mode. Defaults to tts.' },
|
|
105
107
|
model: { type: 'string', description: 'One of the configured audio models/voices. Defaults to the first configured model.' },
|
|
106
108
|
voice: { type: 'string', description: 'Optional voice id/name for TTS providers.' },
|
|
109
|
+
preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
|
|
107
110
|
speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported.' },
|
|
108
111
|
duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
|
|
109
112
|
format: { type: 'string', description: 'Output format such as mp3 or wav.' },
|
|
@@ -117,15 +120,24 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
117
120
|
async execute(args, exec) {
|
|
118
121
|
const config = resolve()
|
|
119
122
|
ensureConfigured(config)
|
|
120
|
-
const
|
|
123
|
+
const mode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
|
|
124
|
+
const picked = mode === 'voice_design'
|
|
125
|
+
? (() => {
|
|
126
|
+
const usable = config.channels.filter(channel => channel.apiUrl.trim() !== '' && channel.apiKey.trim() !== '')
|
|
127
|
+
const target = usable.find(channel => channel.id === config.defaultChannelId) ?? usable[0]
|
|
128
|
+
if (target === undefined) throw new AudioGenError('No usable audio channel is configured for voice design.', 'no-channel-available')
|
|
129
|
+
return { channel: target, alias: '', upstream: '' }
|
|
130
|
+
})()
|
|
131
|
+
: resolveModel(config, args.model)
|
|
121
132
|
const request: GenerateAudioRequest = {
|
|
122
|
-
mode
|
|
133
|
+
mode,
|
|
123
134
|
model: picked.alias,
|
|
124
135
|
upstream: picked.upstream,
|
|
125
136
|
channelId: picked.channel.id,
|
|
126
137
|
channel: picked.channel.name,
|
|
127
138
|
prompt: args.prompt.trim(),
|
|
128
139
|
...(typeof args.voice === 'string' && args.voice.trim() !== '' ? { voice: args.voice.trim() } : {}),
|
|
140
|
+
...(typeof args.preview_text === 'string' && args.preview_text.trim() !== '' ? { previewText: args.preview_text.trim() } : {}),
|
|
129
141
|
...(typeof args.speed === 'number' ? { speed: args.speed } : {}),
|
|
130
142
|
...(typeof args.duration === 'number' ? { duration: args.duration } : {}),
|
|
131
143
|
...(typeof args.format === 'string' && args.format.trim() !== '' ? { format: args.format.trim() } : {}),
|
|
@@ -140,6 +152,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
140
152
|
url: `/api/dsh-audiogen/audio/${encodeURIComponent(saved.file)}`,
|
|
141
153
|
mime: saved.mime,
|
|
142
154
|
bytes: saved.bytes,
|
|
155
|
+
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
143
156
|
})
|
|
144
157
|
}
|
|
145
158
|
try {
|
|
@@ -159,6 +172,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
159
172
|
mime: audio[index]!.mime,
|
|
160
173
|
bytes: audio[index]!.bytes,
|
|
161
174
|
url: audio[index]!.url,
|
|
175
|
+
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
162
176
|
})),
|
|
163
177
|
channelId: picked.channel.id,
|
|
164
178
|
channel: picked.channel.name,
|