dsh-audiogen 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/lib/client.js +318 -264
- package/lib/client.js.map +1 -1
- package/lib/index.js +251 -43
- package/package.json +1 -1
- package/src/audio-engine.ts +46 -22
- package/src/audio-models.ts +129 -0
- package/src/audio-presets.ts +24 -15
- package/src/client/AudioGenPanel.tsx +11 -7
- package/src/client/SettingsCard.tsx +52 -5
- package/src/client/channels-form.ts +5 -1
- package/src/client/locales.ts +2 -2
- package/src/client/settings-scope.ts +10 -4
- package/src/protocol.ts +22 -1
- package/src/routes.ts +28 -1
package/lib/index.js
CHANGED
|
@@ -22,6 +22,8 @@ const SETTINGS_API = {
|
|
|
22
22
|
const GENERATE_API = "/api/dsh-audiogen/generate";
|
|
23
23
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
24
24
|
const PRESETS_API = "/api/dsh-audiogen/presets";
|
|
25
|
+
/** Host-mediated model/voice discovery endpoint. */
|
|
26
|
+
const MODEL_API = { discover: "/api/dsh-audiogen/models/discover" };
|
|
25
27
|
/** Loopback-only audio file reader for panel/tool-result previews. */
|
|
26
28
|
const AUDIO_API = { file: "/api/dsh-audiogen/audio" };
|
|
27
29
|
/** Host-persisted generation history routes. */
|
|
@@ -91,7 +93,7 @@ function isOpenAICompatible(channel, mode) {
|
|
|
91
93
|
function isElevenLabs(channel) {
|
|
92
94
|
return isPreset(channel, "elevenlabs") || /elevenlabs/i.test(channel.apiUrl);
|
|
93
95
|
}
|
|
94
|
-
function isMiniMax(channel) {
|
|
96
|
+
function isMiniMax$1(channel) {
|
|
95
97
|
return isPreset(channel, "minimax") || /minimax/i.test(channel.apiUrl);
|
|
96
98
|
}
|
|
97
99
|
function isStability(channel) {
|
|
@@ -183,13 +185,14 @@ async function normalizeAudioResponse(response, options) {
|
|
|
183
185
|
} catch {
|
|
184
186
|
throw new AudioGenError("audio endpoint returned an unprocessable response body", "audio-bad-response");
|
|
185
187
|
}
|
|
186
|
-
const
|
|
187
|
-
if (
|
|
188
|
+
const encoded = findBase64Audio(parsed);
|
|
189
|
+
if (encoded !== void 0 && encoded.length > 0) {
|
|
188
190
|
let data;
|
|
189
191
|
try {
|
|
190
|
-
|
|
192
|
+
const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0;
|
|
193
|
+
data = new Uint8Array(Buffer.from(encoded, isHex ? "hex" : "base64"));
|
|
191
194
|
} catch {
|
|
192
|
-
throw new AudioGenError("audio endpoint returned invalid
|
|
195
|
+
throw new AudioGenError("audio endpoint returned invalid audio encoding", "audio-bad-response");
|
|
193
196
|
}
|
|
194
197
|
return [{
|
|
195
198
|
data,
|
|
@@ -274,27 +277,47 @@ async function elevenLabs(channel, request, signal) {
|
|
|
274
277
|
fallbackMime: "audio/mpeg"
|
|
275
278
|
});
|
|
276
279
|
}
|
|
280
|
+
function minimaxApiBase(base) {
|
|
281
|
+
const trimmed = endpointBase(base);
|
|
282
|
+
return /\/v1$/i.test(trimmed) ? trimmed : `${trimmed}/v1`;
|
|
283
|
+
}
|
|
277
284
|
async function minimax(channel, request, signal) {
|
|
278
|
-
const base =
|
|
279
|
-
const
|
|
280
|
-
const model = (request.upstream ?? request.model) || "speech-01-turbo";
|
|
285
|
+
const base = minimaxApiBase(channel.apiUrl);
|
|
286
|
+
const model = (request.upstream ?? request.model) || (request.mode === "music" ? "music-3.0" : "speech-2.8-hd");
|
|
281
287
|
const voice = request.voice ?? request.model ?? "";
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
288
|
+
let endpoint;
|
|
289
|
+
let body;
|
|
290
|
+
if (request.mode === "music") {
|
|
291
|
+
endpoint = `${base}/music_generation`;
|
|
292
|
+
body = {
|
|
293
|
+
model,
|
|
294
|
+
prompt: request.prompt,
|
|
295
|
+
...request.duration !== void 0 ? { duration: request.duration } : {},
|
|
296
|
+
audio_setting: {
|
|
297
|
+
format: request.format ?? "mp3",
|
|
298
|
+
sample_rate: 44100,
|
|
299
|
+
bitrate: 256e3
|
|
300
|
+
}
|
|
301
|
+
};
|
|
302
|
+
} else {
|
|
303
|
+
endpoint = `${base}/t2a_v2`;
|
|
304
|
+
body = {
|
|
305
|
+
model,
|
|
306
|
+
text: request.prompt,
|
|
307
|
+
stream: false,
|
|
308
|
+
...voice === "" ? {} : { voice_setting: {
|
|
309
|
+
voice_id: voice,
|
|
310
|
+
...request.speed !== void 0 ? { speed: request.speed } : {},
|
|
311
|
+
vol: 1,
|
|
312
|
+
pitch: 0
|
|
313
|
+
} },
|
|
314
|
+
audio_setting: {
|
|
315
|
+
format: request.format ?? "mp3",
|
|
316
|
+
sample_rate: 32e3,
|
|
317
|
+
bitrate: 128e3
|
|
318
|
+
}
|
|
319
|
+
};
|
|
320
|
+
}
|
|
298
321
|
return normalizeAudioResponse(await fetchWithTimeout(endpoint, {
|
|
299
322
|
method: "POST",
|
|
300
323
|
redirect: "error",
|
|
@@ -370,7 +393,7 @@ async function generateAudio(channel, request, signal) {
|
|
|
370
393
|
if (channel.apiKey.trim() === "") throw new AudioGenError("channel API key is not configured", "audio-no-key");
|
|
371
394
|
if (request.prompt.trim() === "") throw new AudioGenError("audio prompt/text is required", "audio-empty-prompt");
|
|
372
395
|
if (isElevenLabs(channel)) return elevenLabs(channel, request, signal);
|
|
373
|
-
if (isMiniMax(channel)) return minimax(channel, request, signal);
|
|
396
|
+
if (isMiniMax$1(channel)) return minimax(channel, request, signal);
|
|
374
397
|
if (isStability(channel)) return stabilityAudio(channel, request, signal);
|
|
375
398
|
if (isOpenAICompatible(channel, request.mode)) return openAITTS(channel, request, signal);
|
|
376
399
|
return genericAudio(channel, request, signal);
|
|
@@ -386,15 +409,18 @@ const AUDIO_PRESETS = [
|
|
|
386
409
|
models: [
|
|
387
410
|
{
|
|
388
411
|
alias: "tts-1",
|
|
389
|
-
id: "tts-1"
|
|
412
|
+
id: "tts-1",
|
|
413
|
+
category: "tts"
|
|
390
414
|
},
|
|
391
415
|
{
|
|
392
416
|
alias: "tts-1-hd",
|
|
393
|
-
id: "tts-1-hd"
|
|
417
|
+
id: "tts-1-hd",
|
|
418
|
+
category: "tts"
|
|
394
419
|
},
|
|
395
420
|
{
|
|
396
421
|
alias: "gpt-4o-mini-tts",
|
|
397
|
-
id: "gpt-4o-mini-tts"
|
|
422
|
+
id: "gpt-4o-mini-tts",
|
|
423
|
+
category: "tts"
|
|
398
424
|
}
|
|
399
425
|
]
|
|
400
426
|
},
|
|
@@ -406,43 +432,86 @@ const AUDIO_PRESETS = [
|
|
|
406
432
|
models: [
|
|
407
433
|
{
|
|
408
434
|
alias: "Rachel",
|
|
409
|
-
id: "21m00Tcm4TlvDq8ikWAM"
|
|
435
|
+
id: "21m00Tcm4TlvDq8ikWAM",
|
|
436
|
+
category: "tts"
|
|
410
437
|
},
|
|
411
438
|
{
|
|
412
439
|
alias: "Adam",
|
|
413
|
-
id: "pNInz6obpgDQGcFmaJgB"
|
|
440
|
+
id: "pNInz6obpgDQGcFmaJgB",
|
|
441
|
+
category: "tts"
|
|
414
442
|
},
|
|
415
443
|
{
|
|
416
444
|
alias: "Antoni",
|
|
417
|
-
id: "ErXwobaYiN019PkySvjV"
|
|
445
|
+
id: "ErXwobaYiN019PkySvjV",
|
|
446
|
+
category: "tts"
|
|
418
447
|
},
|
|
419
448
|
{
|
|
420
449
|
alias: "Bella",
|
|
421
|
-
id: "EXAVITQu4vr4xnSDxMaL"
|
|
450
|
+
id: "EXAVITQu4vr4xnSDxMaL",
|
|
451
|
+
category: "tts"
|
|
422
452
|
}
|
|
423
453
|
]
|
|
424
454
|
},
|
|
425
455
|
{
|
|
426
456
|
id: "minimax",
|
|
427
457
|
name: "MiniMax",
|
|
428
|
-
apiUrl: "https://api.
|
|
429
|
-
hint: "MiniMax
|
|
458
|
+
apiUrl: "https://api.minimaxi.com",
|
|
459
|
+
hint: "MiniMax 音色设计 / TTS / 音乐生成;可使用“获取可用模型”拉取账号音色",
|
|
430
460
|
models: [
|
|
431
461
|
{
|
|
432
|
-
alias: "speech-
|
|
433
|
-
id: "speech-
|
|
462
|
+
alias: "speech-2.8-hd",
|
|
463
|
+
id: "speech-2.8-hd",
|
|
464
|
+
category: "tts"
|
|
434
465
|
},
|
|
435
466
|
{
|
|
436
|
-
alias: "speech-
|
|
437
|
-
id: "speech-
|
|
467
|
+
alias: "speech-2.8-turbo",
|
|
468
|
+
id: "speech-2.8-turbo",
|
|
469
|
+
category: "tts"
|
|
438
470
|
},
|
|
439
471
|
{
|
|
440
|
-
alias: "speech-
|
|
441
|
-
id: "speech-
|
|
472
|
+
alias: "speech-2.6-hd",
|
|
473
|
+
id: "speech-2.6-hd",
|
|
474
|
+
category: "tts"
|
|
475
|
+
},
|
|
476
|
+
{
|
|
477
|
+
alias: "speech-2.6-turbo",
|
|
478
|
+
id: "speech-2.6-turbo",
|
|
479
|
+
category: "tts"
|
|
442
480
|
},
|
|
443
481
|
{
|
|
444
482
|
alias: "speech-02-hd",
|
|
445
|
-
id: "speech-02-hd"
|
|
483
|
+
id: "speech-02-hd",
|
|
484
|
+
category: "tts"
|
|
485
|
+
},
|
|
486
|
+
{
|
|
487
|
+
alias: "speech-02-turbo",
|
|
488
|
+
id: "speech-02-turbo",
|
|
489
|
+
category: "tts"
|
|
490
|
+
},
|
|
491
|
+
{
|
|
492
|
+
alias: "speech-01-hd",
|
|
493
|
+
id: "speech-01-hd",
|
|
494
|
+
category: "tts"
|
|
495
|
+
},
|
|
496
|
+
{
|
|
497
|
+
alias: "speech-01-turbo",
|
|
498
|
+
id: "speech-01-turbo",
|
|
499
|
+
category: "tts"
|
|
500
|
+
},
|
|
501
|
+
{
|
|
502
|
+
alias: "music-3.0",
|
|
503
|
+
id: "music-3.0",
|
|
504
|
+
category: "music"
|
|
505
|
+
},
|
|
506
|
+
{
|
|
507
|
+
alias: "music-2.6",
|
|
508
|
+
id: "music-2.6",
|
|
509
|
+
category: "music"
|
|
510
|
+
},
|
|
511
|
+
{
|
|
512
|
+
alias: "music-cover",
|
|
513
|
+
id: "music-cover",
|
|
514
|
+
category: "music"
|
|
446
515
|
}
|
|
447
516
|
]
|
|
448
517
|
},
|
|
@@ -453,10 +522,12 @@ const AUDIO_PRESETS = [
|
|
|
453
522
|
hint: "Stability AI 音乐/音效生成(stable-audio 系列)",
|
|
454
523
|
models: [{
|
|
455
524
|
alias: "stable-audio-2.0",
|
|
456
|
-
id: "stable-audio-2.0"
|
|
525
|
+
id: "stable-audio-2.0",
|
|
526
|
+
category: "music"
|
|
457
527
|
}, {
|
|
458
528
|
alias: "stable-audio-1.0",
|
|
459
|
-
id: "stable-audio-1.0"
|
|
529
|
+
id: "stable-audio-1.0",
|
|
530
|
+
category: "music"
|
|
460
531
|
}]
|
|
461
532
|
},
|
|
462
533
|
{
|
|
@@ -472,6 +543,113 @@ function audioPresetById(id) {
|
|
|
472
543
|
return AUDIO_PRESETS.find((preset) => preset.id === id);
|
|
473
544
|
}
|
|
474
545
|
//#endregion
|
|
546
|
+
//#region src/audio-models.ts
|
|
547
|
+
function isMiniMax(channel) {
|
|
548
|
+
return channel.preset === "minimax" || /minimax/i.test(channel.apiUrl);
|
|
549
|
+
}
|
|
550
|
+
function baseUrl(url) {
|
|
551
|
+
return url.trim().replace(/\/+$/, "");
|
|
552
|
+
}
|
|
553
|
+
function categoryFor(id) {
|
|
554
|
+
const value = id.toLowerCase();
|
|
555
|
+
if (/(tts|speech|voice|t2a)/i.test(value)) return "tts";
|
|
556
|
+
if (/(music|song|cover|lyrics)/i.test(value)) return "music";
|
|
557
|
+
if (/(sfx|sound.?effect|effect|foley)/i.test(value)) return "sfx";
|
|
558
|
+
}
|
|
559
|
+
async function postJson(url, apiKey, body) {
|
|
560
|
+
const response = await fetch(url, {
|
|
561
|
+
method: "POST",
|
|
562
|
+
headers: {
|
|
563
|
+
authorization: `Bearer ${apiKey.trim()}`,
|
|
564
|
+
"content-type": "application/json"
|
|
565
|
+
},
|
|
566
|
+
body: JSON.stringify(body)
|
|
567
|
+
});
|
|
568
|
+
if (!response.ok) {
|
|
569
|
+
const text = await response.text().catch(() => "");
|
|
570
|
+
throw new Error(`HTTP ${response.status}${text === "" ? "" : `: ${text.slice(0, 300)}`}`);
|
|
571
|
+
}
|
|
572
|
+
return response.json();
|
|
573
|
+
}
|
|
574
|
+
/** Discover available models/voices for a channel. */
|
|
575
|
+
async function discoverAudioModels(channel) {
|
|
576
|
+
if (channel.apiUrl.trim() === "") throw new Error("API URL is not configured");
|
|
577
|
+
if (channel.apiKey.trim() === "") throw new Error("API key is not configured");
|
|
578
|
+
if (isMiniMax(channel)) {
|
|
579
|
+
const payload = await postJson(`${baseUrl(channel.apiUrl).replace(/\/v1$/i, "")}/v1/get_voice`, channel.apiKey, { voice_type: "all" });
|
|
580
|
+
if (payload.base_resp?.status_code !== void 0 && payload.base_resp.status_code !== 0) throw new Error(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`);
|
|
581
|
+
const models = [];
|
|
582
|
+
for (const voice of payload.system_voice ?? []) {
|
|
583
|
+
const id = voice.voice_id?.trim() ?? "";
|
|
584
|
+
if (id === "") continue;
|
|
585
|
+
models.push({
|
|
586
|
+
alias: voice.voice_name?.trim() || id,
|
|
587
|
+
id,
|
|
588
|
+
category: "tts",
|
|
589
|
+
...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
|
|
590
|
+
});
|
|
591
|
+
}
|
|
592
|
+
for (const voice of payload.voice_cloning ?? []) {
|
|
593
|
+
const id = voice.voice_id?.trim() ?? "";
|
|
594
|
+
if (id === "") continue;
|
|
595
|
+
models.push({
|
|
596
|
+
alias: id,
|
|
597
|
+
id,
|
|
598
|
+
category: "tts",
|
|
599
|
+
...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
|
|
600
|
+
});
|
|
601
|
+
}
|
|
602
|
+
for (const voice of payload.voice_generation ?? []) {
|
|
603
|
+
const id = voice.voice_id?.trim() ?? "";
|
|
604
|
+
if (id === "") continue;
|
|
605
|
+
models.push({
|
|
606
|
+
alias: id,
|
|
607
|
+
id,
|
|
608
|
+
category: "tts",
|
|
609
|
+
...voice.description !== void 0 && voice.description.length > 0 ? { description: voice.description.join(";") } : {}
|
|
610
|
+
});
|
|
611
|
+
}
|
|
612
|
+
const music = (audioPresetById("minimax")?.models ?? []).filter((model) => model.category === "music");
|
|
613
|
+
for (const model of music) models.push({
|
|
614
|
+
...model,
|
|
615
|
+
category: "music"
|
|
616
|
+
});
|
|
617
|
+
return {
|
|
618
|
+
models: dedupe(models),
|
|
619
|
+
source: "MiniMax get_voice + built-in music catalog"
|
|
620
|
+
};
|
|
621
|
+
}
|
|
622
|
+
const url = `${baseUrl(channel.apiUrl)}/models`;
|
|
623
|
+
const response = await fetch(url, { headers: { authorization: `Bearer ${channel.apiKey.trim()}` } });
|
|
624
|
+
if (!response.ok) throw new Error(`model list request failed (HTTP ${response.status}); please add models manually`);
|
|
625
|
+
const payload = await response.json();
|
|
626
|
+
const models = [];
|
|
627
|
+
for (const item of payload.data ?? []) {
|
|
628
|
+
const id = item.id?.trim() ?? "";
|
|
629
|
+
if (id === "") continue;
|
|
630
|
+
const category = categoryFor(id) ?? "tts";
|
|
631
|
+
models.push({
|
|
632
|
+
alias: id,
|
|
633
|
+
id,
|
|
634
|
+
category
|
|
635
|
+
});
|
|
636
|
+
}
|
|
637
|
+
return {
|
|
638
|
+
models: dedupe(models),
|
|
639
|
+
source: "OpenAI-compatible /models"
|
|
640
|
+
};
|
|
641
|
+
}
|
|
642
|
+
function dedupe(models) {
|
|
643
|
+
const seen = /* @__PURE__ */ new Set();
|
|
644
|
+
const out = [];
|
|
645
|
+
for (const model of models) {
|
|
646
|
+
if (seen.has(model.id)) continue;
|
|
647
|
+
seen.add(model.id);
|
|
648
|
+
out.push(model);
|
|
649
|
+
}
|
|
650
|
+
return out;
|
|
651
|
+
}
|
|
652
|
+
//#endregion
|
|
475
653
|
//#region src/audio-store.ts
|
|
476
654
|
/**
|
|
477
655
|
* Host-side persistence for generated audio and generation history.
|
|
@@ -758,6 +936,36 @@ function makeRoutes(deps) {
|
|
|
758
936
|
});
|
|
759
937
|
}
|
|
760
938
|
},
|
|
939
|
+
{
|
|
940
|
+
kind: "exact",
|
|
941
|
+
path: MODEL_API.discover,
|
|
942
|
+
handler: async (req, res) => {
|
|
943
|
+
if (!guard(req, res, "POST")) return;
|
|
944
|
+
const body = await readJsonBody(req);
|
|
945
|
+
const view = deps.resolveChannels();
|
|
946
|
+
const stored = view.channels.find((candidate) => candidate.id === (typeof body?.channelId === "string" ? body.channelId : void 0)) ?? view.channels.find((candidate) => candidate.id === view.defaultChannelId) ?? view.channels[0];
|
|
947
|
+
const channel = {
|
|
948
|
+
id: stored?.id ?? "preview",
|
|
949
|
+
preset: stored?.preset ?? "",
|
|
950
|
+
name: stored?.name ?? "",
|
|
951
|
+
apiUrl: typeof body?.apiUrl === "string" && body.apiUrl.trim() !== "" ? body.apiUrl.trim() : stored?.apiUrl ?? "",
|
|
952
|
+
apiKey: typeof body?.apiKey === "string" && body.apiKey.trim() !== "" ? body.apiKey.trim() : stored?.apiKey ?? "",
|
|
953
|
+
models: stored?.models ?? []
|
|
954
|
+
};
|
|
955
|
+
try {
|
|
956
|
+
writeJson(res, 200, {
|
|
957
|
+
ok: true,
|
|
958
|
+
...await discoverAudioModels(channel)
|
|
959
|
+
});
|
|
960
|
+
} catch (error) {
|
|
961
|
+
writeJson(res, 200, {
|
|
962
|
+
ok: false,
|
|
963
|
+
code: "model-discovery-failed",
|
|
964
|
+
message: messageOf(error)
|
|
965
|
+
});
|
|
966
|
+
}
|
|
967
|
+
}
|
|
968
|
+
},
|
|
761
969
|
{
|
|
762
970
|
kind: "exact",
|
|
763
971
|
path: SETTINGS_API.describe,
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-audiogen",
|
|
3
3
|
"description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.2.0",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|
package/src/audio-engine.ts
CHANGED
|
@@ -187,13 +187,14 @@ async function normalizeAudioResponse(
|
|
|
187
187
|
} catch {
|
|
188
188
|
throw new AudioGenError('audio endpoint returned an unprocessable response body', 'audio-bad-response')
|
|
189
189
|
}
|
|
190
|
-
const
|
|
191
|
-
if (
|
|
190
|
+
const encoded = findBase64Audio(parsed)
|
|
191
|
+
if (encoded !== undefined && encoded.length > 0) {
|
|
192
192
|
let data: Uint8Array
|
|
193
193
|
try {
|
|
194
|
-
|
|
194
|
+
const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0
|
|
195
|
+
data = new Uint8Array(Buffer.from(encoded, isHex ? 'hex' : 'base64'))
|
|
195
196
|
} catch {
|
|
196
|
-
throw new AudioGenError('audio endpoint returned invalid
|
|
197
|
+
throw new AudioGenError('audio endpoint returned invalid audio encoding', 'audio-bad-response')
|
|
197
198
|
}
|
|
198
199
|
return [{ data, mime: detectAudioMime(data) ?? contentType ?? 'audio/mpeg' }]
|
|
199
200
|
}
|
|
@@ -268,27 +269,50 @@ async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest,
|
|
|
268
269
|
return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
269
270
|
}
|
|
270
271
|
|
|
272
|
+
function minimaxApiBase(base: string): string {
|
|
273
|
+
const trimmed = endpointBase(base)
|
|
274
|
+
return /\/v1$/i.test(trimmed) ? trimmed : `${trimmed}/v1`
|
|
275
|
+
}
|
|
276
|
+
|
|
271
277
|
async function minimax(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string }>> {
|
|
272
|
-
const base =
|
|
273
|
-
const
|
|
274
|
-
const model = (request.upstream ?? request.model) || 'speech-01-turbo'
|
|
278
|
+
const base = minimaxApiBase(channel.apiUrl)
|
|
279
|
+
const model = (request.upstream ?? request.model) || (request.mode === 'music' ? 'music-3.0' : 'speech-2.8-hd')
|
|
275
280
|
const voice = request.voice ?? request.model ?? ''
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
}
|
|
281
|
+
|
|
282
|
+
let endpoint: string
|
|
283
|
+
let body: Record<string, unknown>
|
|
284
|
+
if (request.mode === 'music') {
|
|
285
|
+
endpoint = `${base}/music_generation`
|
|
286
|
+
body = {
|
|
287
|
+
model,
|
|
288
|
+
prompt: request.prompt,
|
|
289
|
+
...(request.duration !== undefined ? { duration: request.duration } : {}),
|
|
290
|
+
audio_setting: {
|
|
291
|
+
format: request.format ?? 'mp3',
|
|
292
|
+
sample_rate: 44100,
|
|
293
|
+
bitrate: 256000,
|
|
294
|
+
},
|
|
295
|
+
}
|
|
296
|
+
} else {
|
|
297
|
+
endpoint = `${base}/t2a_v2`
|
|
298
|
+
body = {
|
|
299
|
+
model,
|
|
300
|
+
text: request.prompt,
|
|
301
|
+
stream: false,
|
|
302
|
+
...(voice === '' ? {} : { voice_setting: {
|
|
303
|
+
voice_id: voice,
|
|
304
|
+
...(request.speed !== undefined ? { speed: request.speed } : {}),
|
|
305
|
+
vol: 1,
|
|
306
|
+
pitch: 0,
|
|
307
|
+
} }),
|
|
308
|
+
audio_setting: {
|
|
309
|
+
format: request.format ?? 'mp3',
|
|
310
|
+
sample_rate: 32000,
|
|
311
|
+
bitrate: 128000,
|
|
312
|
+
},
|
|
313
|
+
}
|
|
291
314
|
}
|
|
315
|
+
|
|
292
316
|
const response = await fetchWithTimeout(endpoint, {
|
|
293
317
|
method: 'POST',
|
|
294
318
|
redirect: 'error',
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Host-side model/voice discovery.
|
|
3
|
+
*
|
|
4
|
+
* MiniMax exposes a voice-management API that returns all available system and
|
|
5
|
+
* user-generated voice ids; we combine those with the known MiniMax music
|
|
6
|
+
* models so the settings card can offer a full categorized catalog.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { AudioChannel } from './audio-engine.ts'
|
|
10
|
+
import type { AudioModelCategory, DiscoveredAudioModel } from './protocol.ts'
|
|
11
|
+
import { audioPresetById } from './audio-presets.ts'
|
|
12
|
+
|
|
13
|
+
function isMiniMax(channel: AudioChannel): boolean {
|
|
14
|
+
return channel.preset === 'minimax' || /minimax/i.test(channel.apiUrl)
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function baseUrl(url: string): string {
|
|
18
|
+
return url.trim().replace(/\/+$/, '')
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
function categoryFor(id: string): AudioModelCategory | undefined {
|
|
22
|
+
const value = id.toLowerCase()
|
|
23
|
+
if (/(tts|speech|voice|t2a)/i.test(value)) return 'tts'
|
|
24
|
+
if (/(music|song|cover|lyrics)/i.test(value)) return 'music'
|
|
25
|
+
if (/(sfx|sound.?effect|effect|foley)/i.test(value)) return 'sfx'
|
|
26
|
+
return undefined
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
async function postJson(url: string, apiKey: string, body: unknown): Promise<unknown> {
|
|
30
|
+
const response = await fetch(url, {
|
|
31
|
+
method: 'POST',
|
|
32
|
+
headers: {
|
|
33
|
+
authorization: `Bearer ${apiKey.trim()}`,
|
|
34
|
+
'content-type': 'application/json',
|
|
35
|
+
},
|
|
36
|
+
body: JSON.stringify(body),
|
|
37
|
+
})
|
|
38
|
+
if (!response.ok) {
|
|
39
|
+
const text = await response.text().catch(() => '')
|
|
40
|
+
throw new Error(`HTTP ${response.status}${text === '' ? '' : `: ${text.slice(0, 300)}`}`)
|
|
41
|
+
}
|
|
42
|
+
return response.json()
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** Discover available models/voices for a channel. */
|
|
46
|
+
export async function discoverAudioModels(channel: AudioChannel): Promise<{ models: DiscoveredAudioModel[]; source: string }> {
|
|
47
|
+
if (channel.apiUrl.trim() === '') throw new Error('API URL is not configured')
|
|
48
|
+
if (channel.apiKey.trim() === '') throw new Error('API key is not configured')
|
|
49
|
+
|
|
50
|
+
if (isMiniMax(channel)) {
|
|
51
|
+
const base = baseUrl(channel.apiUrl).replace(/\/v1$/i, '')
|
|
52
|
+
const url = `${base}/v1/get_voice`
|
|
53
|
+
const payload = await postJson(url, channel.apiKey, { voice_type: 'all' }) as {
|
|
54
|
+
system_voice?: Array<{ voice_id?: string; voice_name?: string; description?: string[] }>
|
|
55
|
+
voice_cloning?: Array<{ voice_id?: string; description?: string[] }>
|
|
56
|
+
voice_generation?: Array<{ voice_id?: string; description?: string[] }>
|
|
57
|
+
base_resp?: { status_code?: number; status_msg?: string }
|
|
58
|
+
}
|
|
59
|
+
if (payload.base_resp?.status_code !== undefined && payload.base_resp.status_code !== 0) {
|
|
60
|
+
throw new Error(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`)
|
|
61
|
+
}
|
|
62
|
+
const models: DiscoveredAudioModel[] = []
|
|
63
|
+
for (const voice of payload.system_voice ?? []) {
|
|
64
|
+
const id = voice.voice_id?.trim() ?? ''
|
|
65
|
+
if (id === '') continue
|
|
66
|
+
models.push({
|
|
67
|
+
alias: voice.voice_name?.trim() || id,
|
|
68
|
+
id,
|
|
69
|
+
category: 'tts',
|
|
70
|
+
...(voice.description !== undefined && voice.description.length > 0 ? { description: voice.description.join(';') } : {}),
|
|
71
|
+
})
|
|
72
|
+
}
|
|
73
|
+
for (const voice of payload.voice_cloning ?? []) {
|
|
74
|
+
const id = voice.voice_id?.trim() ?? ''
|
|
75
|
+
if (id === '') continue
|
|
76
|
+
models.push({
|
|
77
|
+
alias: id,
|
|
78
|
+
id,
|
|
79
|
+
category: 'tts',
|
|
80
|
+
...(voice.description !== undefined && voice.description.length > 0 ? { description: voice.description.join(';') } : {}),
|
|
81
|
+
})
|
|
82
|
+
}
|
|
83
|
+
for (const voice of payload.voice_generation ?? []) {
|
|
84
|
+
const id = voice.voice_id?.trim() ?? ''
|
|
85
|
+
if (id === '') continue
|
|
86
|
+
models.push({
|
|
87
|
+
alias: id,
|
|
88
|
+
id,
|
|
89
|
+
category: 'tts',
|
|
90
|
+
...(voice.description !== undefined && voice.description.length > 0 ? { description: voice.description.join(';') } : {}),
|
|
91
|
+
})
|
|
92
|
+
}
|
|
93
|
+
// MiniMax music models are not returned by get_voice; append the static catalog.
|
|
94
|
+
const music = (audioPresetById('minimax')?.models ?? []).filter(model => model.category === 'music')
|
|
95
|
+
for (const model of music) models.push({ ...model, category: 'music' as const })
|
|
96
|
+
const deduped = dedupe(models)
|
|
97
|
+
return { models: deduped, source: 'MiniMax get_voice + built-in music catalog' }
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// Best-effort OpenAI-compatible /models discovery.
|
|
101
|
+
const base = baseUrl(channel.apiUrl)
|
|
102
|
+
const url = `${base}/models`
|
|
103
|
+
const response = await fetch(url, {
|
|
104
|
+
headers: { authorization: `Bearer ${channel.apiKey.trim()}` },
|
|
105
|
+
})
|
|
106
|
+
if (!response.ok) {
|
|
107
|
+
throw new Error(`model list request failed (HTTP ${response.status}); please add models manually`)
|
|
108
|
+
}
|
|
109
|
+
const payload = await response.json() as { data?: Array<{ id?: string }> }
|
|
110
|
+
const models: DiscoveredAudioModel[] = []
|
|
111
|
+
for (const item of payload.data ?? []) {
|
|
112
|
+
const id = item.id?.trim() ?? ''
|
|
113
|
+
if (id === '') continue
|
|
114
|
+
const category = categoryFor(id) ?? 'tts'
|
|
115
|
+
models.push({ alias: id, id, category })
|
|
116
|
+
}
|
|
117
|
+
return { models: dedupe(models), source: 'OpenAI-compatible /models' }
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function dedupe(models: DiscoveredAudioModel[]): DiscoveredAudioModel[] {
|
|
121
|
+
const seen = new Set<string>()
|
|
122
|
+
const out: DiscoveredAudioModel[] = []
|
|
123
|
+
for (const model of models) {
|
|
124
|
+
if (seen.has(model.id)) continue
|
|
125
|
+
seen.add(model.id)
|
|
126
|
+
out.push(model)
|
|
127
|
+
}
|
|
128
|
+
return out
|
|
129
|
+
}
|