dsh-audiogen 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/lib/client.js +373 -291
- package/lib/client.js.map +1 -1
- package/lib/index.js +333 -53
- package/package.json +1 -1
- package/src/agent-audio-tools.ts +19 -5
- package/src/audio-engine.ts +96 -29
- package/src/audio-models.ts +129 -0
- package/src/audio-presets.ts +24 -15
- package/src/audio-store.ts +2 -0
- package/src/client/AudioGenPanel.tsx +76 -36
- package/src/client/SettingsCard.tsx +52 -5
- package/src/client/channels-form.ts +5 -1
- package/src/client/locales.ts +6 -4
- package/src/client/settings-scope.ts +10 -4
- package/src/protocol.ts +31 -2
- package/src/routes.ts +35 -2
package/src/audio-engine.ts
CHANGED
|
@@ -164,7 +164,7 @@ async function fetchWithTimeout(url: string, init: RequestInit, timeoutMs: numbe
|
|
|
164
164
|
async function normalizeAudioResponse(
|
|
165
165
|
response: Response,
|
|
166
166
|
options: { apiKey: string; fallbackMime?: string },
|
|
167
|
-
): Promise<Array<{ data: Uint8Array; mime: string }>> {
|
|
167
|
+
): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
168
168
|
if (!response.ok) {
|
|
169
169
|
let detail = ''
|
|
170
170
|
try {
|
|
@@ -187,13 +187,14 @@ async function normalizeAudioResponse(
|
|
|
187
187
|
} catch {
|
|
188
188
|
throw new AudioGenError('audio endpoint returned an unprocessable response body', 'audio-bad-response')
|
|
189
189
|
}
|
|
190
|
-
const
|
|
191
|
-
if (
|
|
190
|
+
const encoded = findBase64Audio(parsed)
|
|
191
|
+
if (encoded !== undefined && encoded.length > 0) {
|
|
192
192
|
let data: Uint8Array
|
|
193
193
|
try {
|
|
194
|
-
|
|
194
|
+
const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0
|
|
195
|
+
data = new Uint8Array(Buffer.from(encoded, isHex ? 'hex' : 'base64'))
|
|
195
196
|
} catch {
|
|
196
|
-
throw new AudioGenError('audio endpoint returned invalid
|
|
197
|
+
throw new AudioGenError('audio endpoint returned invalid audio encoding', 'audio-bad-response')
|
|
197
198
|
}
|
|
198
199
|
return [{ data, mime: detectAudioMime(data) ?? contentType ?? 'audio/mpeg' }]
|
|
199
200
|
}
|
|
@@ -212,7 +213,7 @@ async function normalizeAudioResponse(
|
|
|
212
213
|
return [{ data: buffer, mime: audioMime(buffer, response.headers.get('content-type') ?? contentType ?? null) }]
|
|
213
214
|
}
|
|
214
215
|
|
|
215
|
-
async function openAITTS(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string }>> {
|
|
216
|
+
async function openAITTS(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
216
217
|
const base = endpointBase(channel.apiUrl)
|
|
217
218
|
const endpoint = /\/audio\/speech(\?|$)/i.test(base) ? base : `${base}/audio/speech`
|
|
218
219
|
const model = (request.upstream ?? request.model) || 'tts-1'
|
|
@@ -238,7 +239,7 @@ async function openAITTS(channel: AudioChannel, request: GenerateAudioRequest, s
|
|
|
238
239
|
return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
239
240
|
}
|
|
240
241
|
|
|
241
|
-
async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string }>> {
|
|
242
|
+
async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
242
243
|
const base = endpointBase(channel.apiUrl)
|
|
243
244
|
const model = (request.upstream ?? request.model) || 'eleven_multilingual_v2'
|
|
244
245
|
const voiceId = (request.voice ?? request.model ?? model).trim()
|
|
@@ -268,27 +269,90 @@ async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest,
|
|
|
268
269
|
return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
269
270
|
}
|
|
270
271
|
|
|
271
|
-
|
|
272
|
-
const
|
|
273
|
-
|
|
274
|
-
|
|
272
|
+
function minimaxApiBase(base: string): string {
|
|
273
|
+
const trimmed = endpointBase(base)
|
|
274
|
+
return /\/v1$/i.test(trimmed) ? trimmed : `${trimmed}/v1`
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
async function minimax(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
278
|
+
const base = minimaxApiBase(channel.apiUrl)
|
|
279
|
+
const model = (request.upstream ?? request.model) || (request.mode === 'music' ? 'music-3.0' : 'speech-2.8-hd')
|
|
275
280
|
const voice = request.voice ?? request.model ?? ''
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
281
|
+
|
|
282
|
+
if (request.mode === 'voice_design') {
|
|
283
|
+
const endpoint = `${base}/voice_design`
|
|
284
|
+
const body: Record<string, unknown> = {
|
|
285
|
+
prompt: request.prompt,
|
|
286
|
+
preview_text: request.previewText ?? request.voice ?? '你好,这是新设计的音色试听。',
|
|
287
|
+
}
|
|
288
|
+
const response = await fetchWithTimeout(endpoint, {
|
|
289
|
+
method: 'POST',
|
|
290
|
+
redirect: 'error',
|
|
291
|
+
headers: {
|
|
292
|
+
authorization: `Bearer ${channel.apiKey.trim()}`,
|
|
293
|
+
'content-type': 'application/json',
|
|
294
|
+
accept: 'application/json',
|
|
295
|
+
},
|
|
296
|
+
body: JSON.stringify(body),
|
|
297
|
+
signal,
|
|
298
|
+
}, UPSTREAM_TIMEOUT_MS)
|
|
299
|
+
if (!response.ok) {
|
|
300
|
+
const detail = await response.text().catch(() => '')
|
|
301
|
+
throw new AudioGenError(`MiniMax voice design API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
|
|
302
|
+
}
|
|
303
|
+
const payload = await response.json() as {
|
|
304
|
+
voice_id?: string
|
|
305
|
+
trial_audio?: string
|
|
306
|
+
base_resp?: { status_code?: number; status_msg?: string }
|
|
307
|
+
}
|
|
308
|
+
if (payload.base_resp?.status_code !== undefined && payload.base_resp.status_code !== 0) {
|
|
309
|
+
throw new AudioGenError(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`, 'audio-api-error')
|
|
310
|
+
}
|
|
311
|
+
const encoded = payload.trial_audio ?? ''
|
|
312
|
+
if (encoded === '') throw new AudioGenError('MiniMax voice design returned no trial audio', 'audio-empty-result')
|
|
313
|
+
const isHex = /^[0-9a-fA-F]+$/.test(encoded) && encoded.length % 2 === 0
|
|
314
|
+
const data = new Uint8Array(Buffer.from(encoded, isHex ? 'hex' : 'base64'))
|
|
315
|
+
return [{
|
|
316
|
+
data,
|
|
317
|
+
mime: 'audio/mpeg',
|
|
318
|
+
...(payload.voice_id === undefined ? {} : { voiceId: payload.voice_id }),
|
|
319
|
+
}]
|
|
291
320
|
}
|
|
321
|
+
|
|
322
|
+
let endpoint: string
|
|
323
|
+
let body: Record<string, unknown>
|
|
324
|
+
if (request.mode === 'music') {
|
|
325
|
+
endpoint = `${base}/music_generation`
|
|
326
|
+
body = {
|
|
327
|
+
model,
|
|
328
|
+
prompt: request.prompt,
|
|
329
|
+
...(request.duration !== undefined ? { duration: request.duration } : {}),
|
|
330
|
+
audio_setting: {
|
|
331
|
+
format: request.format ?? 'mp3',
|
|
332
|
+
sample_rate: 44100,
|
|
333
|
+
bitrate: 256000,
|
|
334
|
+
},
|
|
335
|
+
}
|
|
336
|
+
} else {
|
|
337
|
+
endpoint = `${base}/t2a_v2`
|
|
338
|
+
body = {
|
|
339
|
+
model,
|
|
340
|
+
text: request.prompt,
|
|
341
|
+
stream: false,
|
|
342
|
+
...(voice === '' ? {} : { voice_setting: {
|
|
343
|
+
voice_id: voice,
|
|
344
|
+
...(request.speed !== undefined ? { speed: request.speed } : {}),
|
|
345
|
+
vol: 1,
|
|
346
|
+
pitch: 0,
|
|
347
|
+
} }),
|
|
348
|
+
audio_setting: {
|
|
349
|
+
format: request.format ?? 'mp3',
|
|
350
|
+
sample_rate: 32000,
|
|
351
|
+
bitrate: 128000,
|
|
352
|
+
},
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
292
356
|
const response = await fetchWithTimeout(endpoint, {
|
|
293
357
|
method: 'POST',
|
|
294
358
|
redirect: 'error',
|
|
@@ -303,7 +367,7 @@ async function minimax(channel: AudioChannel, request: GenerateAudioRequest, sig
|
|
|
303
367
|
return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
304
368
|
}
|
|
305
369
|
|
|
306
|
-
async function stabilityAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string }>> {
|
|
370
|
+
async function stabilityAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
307
371
|
const base = endpointBase(channel.apiUrl)
|
|
308
372
|
const endpoint = /\/generation(\?|$)/i.test(base) ? base : `${base}/generation`
|
|
309
373
|
const model = (request.upstream ?? request.model) || 'stable-audio-2.0'
|
|
@@ -327,7 +391,7 @@ async function stabilityAudio(channel: AudioChannel, request: GenerateAudioReque
|
|
|
327
391
|
return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
328
392
|
}
|
|
329
393
|
|
|
330
|
-
async function genericAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string }>> {
|
|
394
|
+
async function genericAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
331
395
|
const base = endpointBase(channel.apiUrl)
|
|
332
396
|
if (request.mode === 'tts' && !/\/generate(\?|$)/i.test(base)) {
|
|
333
397
|
return openAITTS(channel, request, signal)
|
|
@@ -364,10 +428,13 @@ export async function generateAudio(
|
|
|
364
428
|
channel: AudioChannel,
|
|
365
429
|
request: GenerateAudioRequest,
|
|
366
430
|
signal?: AbortSignal,
|
|
367
|
-
): Promise<Array<{ data: Uint8Array; mime: string }>> {
|
|
431
|
+
): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
368
432
|
if (channel.apiUrl.trim() === '') throw new AudioGenError('channel API URL is not configured', 'audio-no-endpoint')
|
|
369
433
|
if (channel.apiKey.trim() === '') throw new AudioGenError('channel API key is not configured', 'audio-no-key')
|
|
370
434
|
if (request.prompt.trim() === '') throw new AudioGenError('audio prompt/text is required', 'audio-empty-prompt')
|
|
435
|
+
if (request.mode === 'voice_design' && !isMiniMax(channel)) {
|
|
436
|
+
throw new AudioGenError('音色设计当前仅支持 MiniMax 渠道', 'voice-design-unsupported')
|
|
437
|
+
}
|
|
371
438
|
|
|
372
439
|
if (isElevenLabs(channel)) return elevenLabs(channel, request, signal)
|
|
373
440
|
if (isMiniMax(channel)) return minimax(channel, request, signal)
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Host-side model/voice discovery.
|
|
3
|
+
*
|
|
4
|
+
* MiniMax exposes a voice-management API that returns all available system and
|
|
5
|
+
* user-generated voice ids; we combine those with the known MiniMax music
|
|
6
|
+
* models so the settings card can offer a full categorized catalog.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { AudioChannel } from './audio-engine.ts'
|
|
10
|
+
import type { AudioModelCategory, DiscoveredAudioModel } from './protocol.ts'
|
|
11
|
+
import { audioPresetById } from './audio-presets.ts'
|
|
12
|
+
|
|
13
|
+
function isMiniMax(channel: AudioChannel): boolean {
|
|
14
|
+
return channel.preset === 'minimax' || /minimax/i.test(channel.apiUrl)
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function baseUrl(url: string): string {
|
|
18
|
+
return url.trim().replace(/\/+$/, '')
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
function categoryFor(id: string): AudioModelCategory | undefined {
|
|
22
|
+
const value = id.toLowerCase()
|
|
23
|
+
if (/(tts|speech|voice|t2a)/i.test(value)) return 'tts'
|
|
24
|
+
if (/(music|song|cover|lyrics)/i.test(value)) return 'music'
|
|
25
|
+
if (/(sfx|sound.?effect|effect|foley)/i.test(value)) return 'sfx'
|
|
26
|
+
return undefined
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
async function postJson(url: string, apiKey: string, body: unknown): Promise<unknown> {
|
|
30
|
+
const response = await fetch(url, {
|
|
31
|
+
method: 'POST',
|
|
32
|
+
headers: {
|
|
33
|
+
authorization: `Bearer ${apiKey.trim()}`,
|
|
34
|
+
'content-type': 'application/json',
|
|
35
|
+
},
|
|
36
|
+
body: JSON.stringify(body),
|
|
37
|
+
})
|
|
38
|
+
if (!response.ok) {
|
|
39
|
+
const text = await response.text().catch(() => '')
|
|
40
|
+
throw new Error(`HTTP ${response.status}${text === '' ? '' : `: ${text.slice(0, 300)}`}`)
|
|
41
|
+
}
|
|
42
|
+
return response.json()
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** Discover available models/voices for a channel. */
|
|
46
|
+
export async function discoverAudioModels(channel: AudioChannel): Promise<{ models: DiscoveredAudioModel[]; source: string }> {
|
|
47
|
+
if (channel.apiUrl.trim() === '') throw new Error('API URL is not configured')
|
|
48
|
+
if (channel.apiKey.trim() === '') throw new Error('API key is not configured')
|
|
49
|
+
|
|
50
|
+
if (isMiniMax(channel)) {
|
|
51
|
+
const base = baseUrl(channel.apiUrl).replace(/\/v1$/i, '')
|
|
52
|
+
const url = `${base}/v1/get_voice`
|
|
53
|
+
const payload = await postJson(url, channel.apiKey, { voice_type: 'all' }) as {
|
|
54
|
+
system_voice?: Array<{ voice_id?: string; voice_name?: string; description?: string[] }>
|
|
55
|
+
voice_cloning?: Array<{ voice_id?: string; description?: string[] }>
|
|
56
|
+
voice_generation?: Array<{ voice_id?: string; description?: string[] }>
|
|
57
|
+
base_resp?: { status_code?: number; status_msg?: string }
|
|
58
|
+
}
|
|
59
|
+
if (payload.base_resp?.status_code !== undefined && payload.base_resp.status_code !== 0) {
|
|
60
|
+
throw new Error(payload.base_resp.status_msg ?? `MiniMax returned status ${payload.base_resp.status_code}`)
|
|
61
|
+
}
|
|
62
|
+
const models: DiscoveredAudioModel[] = []
|
|
63
|
+
for (const voice of payload.system_voice ?? []) {
|
|
64
|
+
const id = voice.voice_id?.trim() ?? ''
|
|
65
|
+
if (id === '') continue
|
|
66
|
+
models.push({
|
|
67
|
+
alias: voice.voice_name?.trim() || id,
|
|
68
|
+
id,
|
|
69
|
+
category: 'tts',
|
|
70
|
+
...(voice.description !== undefined && voice.description.length > 0 ? { description: voice.description.join(';') } : {}),
|
|
71
|
+
})
|
|
72
|
+
}
|
|
73
|
+
for (const voice of payload.voice_cloning ?? []) {
|
|
74
|
+
const id = voice.voice_id?.trim() ?? ''
|
|
75
|
+
if (id === '') continue
|
|
76
|
+
models.push({
|
|
77
|
+
alias: id,
|
|
78
|
+
id,
|
|
79
|
+
category: 'tts',
|
|
80
|
+
...(voice.description !== undefined && voice.description.length > 0 ? { description: voice.description.join(';') } : {}),
|
|
81
|
+
})
|
|
82
|
+
}
|
|
83
|
+
for (const voice of payload.voice_generation ?? []) {
|
|
84
|
+
const id = voice.voice_id?.trim() ?? ''
|
|
85
|
+
if (id === '') continue
|
|
86
|
+
models.push({
|
|
87
|
+
alias: id,
|
|
88
|
+
id,
|
|
89
|
+
category: 'tts',
|
|
90
|
+
...(voice.description !== undefined && voice.description.length > 0 ? { description: voice.description.join(';') } : {}),
|
|
91
|
+
})
|
|
92
|
+
}
|
|
93
|
+
// MiniMax music models are not returned by get_voice; append the static catalog.
|
|
94
|
+
const music = (audioPresetById('minimax')?.models ?? []).filter(model => model.category === 'music')
|
|
95
|
+
for (const model of music) models.push({ ...model, category: 'music' as const })
|
|
96
|
+
const deduped = dedupe(models)
|
|
97
|
+
return { models: deduped, source: 'MiniMax get_voice + built-in music catalog' }
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// Best-effort OpenAI-compatible /models discovery.
|
|
101
|
+
const base = baseUrl(channel.apiUrl)
|
|
102
|
+
const url = `${base}/models`
|
|
103
|
+
const response = await fetch(url, {
|
|
104
|
+
headers: { authorization: `Bearer ${channel.apiKey.trim()}` },
|
|
105
|
+
})
|
|
106
|
+
if (!response.ok) {
|
|
107
|
+
throw new Error(`model list request failed (HTTP ${response.status}); please add models manually`)
|
|
108
|
+
}
|
|
109
|
+
const payload = await response.json() as { data?: Array<{ id?: string }> }
|
|
110
|
+
const models: DiscoveredAudioModel[] = []
|
|
111
|
+
for (const item of payload.data ?? []) {
|
|
112
|
+
const id = item.id?.trim() ?? ''
|
|
113
|
+
if (id === '') continue
|
|
114
|
+
const category = categoryFor(id) ?? 'tts'
|
|
115
|
+
models.push({ alias: id, id, category })
|
|
116
|
+
}
|
|
117
|
+
return { models: dedupe(models), source: 'OpenAI-compatible /models' }
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function dedupe(models: DiscoveredAudioModel[]): DiscoveredAudioModel[] {
|
|
121
|
+
const seen = new Set<string>()
|
|
122
|
+
const out: DiscoveredAudioModel[] = []
|
|
123
|
+
for (const model of models) {
|
|
124
|
+
if (seen.has(model.id)) continue
|
|
125
|
+
seen.add(model.id)
|
|
126
|
+
out.push(model)
|
|
127
|
+
}
|
|
128
|
+
return out
|
|
129
|
+
}
|
package/src/audio-presets.ts
CHANGED
|
@@ -26,9 +26,9 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
|
|
|
26
26
|
apiUrl: 'https://api.openai.com/v1',
|
|
27
27
|
hint: 'OpenAI 官方语音合成接口(/audio/speech)',
|
|
28
28
|
models: [
|
|
29
|
-
{ alias: 'tts-1', id: 'tts-1' },
|
|
30
|
-
{ alias: 'tts-1-hd', id: 'tts-1-hd' },
|
|
31
|
-
{ alias: 'gpt-4o-mini-tts', id: 'gpt-4o-mini-tts' },
|
|
29
|
+
{ alias: 'tts-1', id: 'tts-1', category: 'tts' },
|
|
30
|
+
{ alias: 'tts-1-hd', id: 'tts-1-hd', category: 'tts' },
|
|
31
|
+
{ alias: 'gpt-4o-mini-tts', id: 'gpt-4o-mini-tts', category: 'tts' },
|
|
32
32
|
],
|
|
33
33
|
},
|
|
34
34
|
{
|
|
@@ -37,22 +37,31 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
|
|
|
37
37
|
apiUrl: 'https://api.elevenlabs.io/v1',
|
|
38
38
|
hint: 'ElevenLabs TTS;模型列表请填写你的 Voice ID(如 Rachel / Adam 等别名)',
|
|
39
39
|
models: [
|
|
40
|
-
{ alias: 'Rachel', id: '21m00Tcm4TlvDq8ikWAM' },
|
|
41
|
-
{ alias: 'Adam', id: 'pNInz6obpgDQGcFmaJgB' },
|
|
42
|
-
{ alias: 'Antoni', id: 'ErXwobaYiN019PkySvjV' },
|
|
43
|
-
{ alias: 'Bella', id: 'EXAVITQu4vr4xnSDxMaL' },
|
|
40
|
+
{ alias: 'Rachel', id: '21m00Tcm4TlvDq8ikWAM', category: 'tts' },
|
|
41
|
+
{ alias: 'Adam', id: 'pNInz6obpgDQGcFmaJgB', category: 'tts' },
|
|
42
|
+
{ alias: 'Antoni', id: 'ErXwobaYiN019PkySvjV', category: 'tts' },
|
|
43
|
+
{ alias: 'Bella', id: 'EXAVITQu4vr4xnSDxMaL', category: 'tts' },
|
|
44
44
|
],
|
|
45
45
|
},
|
|
46
46
|
{
|
|
47
47
|
id: 'minimax',
|
|
48
48
|
name: 'MiniMax',
|
|
49
|
-
apiUrl: 'https://api.
|
|
50
|
-
hint: 'MiniMax
|
|
49
|
+
apiUrl: 'https://api.minimaxi.com',
|
|
50
|
+
hint: 'MiniMax 音色设计 / TTS / 音乐生成;可使用“获取可用模型”拉取账号音色',
|
|
51
51
|
models: [
|
|
52
|
-
|
|
53
|
-
{ alias: 'speech-
|
|
54
|
-
{ alias: 'speech-
|
|
55
|
-
{ alias: 'speech-
|
|
52
|
+
// TTS models
|
|
53
|
+
{ alias: 'speech-2.8-hd', id: 'speech-2.8-hd', category: 'tts' },
|
|
54
|
+
{ alias: 'speech-2.8-turbo', id: 'speech-2.8-turbo', category: 'tts' },
|
|
55
|
+
{ alias: 'speech-2.6-hd', id: 'speech-2.6-hd', category: 'tts' },
|
|
56
|
+
{ alias: 'speech-2.6-turbo', id: 'speech-2.6-turbo', category: 'tts' },
|
|
57
|
+
{ alias: 'speech-02-hd', id: 'speech-02-hd', category: 'tts' },
|
|
58
|
+
{ alias: 'speech-02-turbo', id: 'speech-02-turbo', category: 'tts' },
|
|
59
|
+
{ alias: 'speech-01-hd', id: 'speech-01-hd', category: 'tts' },
|
|
60
|
+
{ alias: 'speech-01-turbo', id: 'speech-01-turbo', category: 'tts' },
|
|
61
|
+
// Music models
|
|
62
|
+
{ alias: 'music-3.0', id: 'music-3.0', category: 'music' },
|
|
63
|
+
{ alias: 'music-2.6', id: 'music-2.6', category: 'music' },
|
|
64
|
+
{ alias: 'music-cover', id: 'music-cover', category: 'music' },
|
|
56
65
|
],
|
|
57
66
|
},
|
|
58
67
|
{
|
|
@@ -61,8 +70,8 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
|
|
|
61
70
|
apiUrl: 'https://api.stability.ai/v2beta/audio',
|
|
62
71
|
hint: 'Stability AI 音乐/音效生成(stable-audio 系列)',
|
|
63
72
|
models: [
|
|
64
|
-
{ alias: 'stable-audio-2.0', id: 'stable-audio-2.0' },
|
|
65
|
-
{ alias: 'stable-audio-1.0', id: 'stable-audio-1.0' },
|
|
73
|
+
{ alias: 'stable-audio-2.0', id: 'stable-audio-2.0', category: 'music' },
|
|
74
|
+
{ alias: 'stable-audio-1.0', id: 'stable-audio-1.0', category: 'music' },
|
|
66
75
|
],
|
|
67
76
|
},
|
|
68
77
|
{
|
package/src/audio-store.ts
CHANGED
|
@@ -99,6 +99,7 @@ export async function appendHistory(entry: HistoryEntryInput): Promise<HistoryEn
|
|
|
99
99
|
model: entry.model,
|
|
100
100
|
prompt: entry.prompt,
|
|
101
101
|
...(entry.voice === undefined ? {} : { voice: entry.voice }),
|
|
102
|
+
...(entry.voiceId === undefined ? {} : { voiceId: entry.voiceId }),
|
|
102
103
|
...(entry.speed === undefined ? {} : { speed: entry.speed }),
|
|
103
104
|
...(entry.duration === undefined ? {} : { duration: entry.duration }),
|
|
104
105
|
...(entry.format === undefined ? {} : { format: entry.format }),
|
|
@@ -106,6 +107,7 @@ export async function appendHistory(entry: HistoryEntryInput): Promise<HistoryEn
|
|
|
106
107
|
url: audio.url,
|
|
107
108
|
mime: audio.mime,
|
|
108
109
|
...(audio.duration === undefined ? {} : { duration: audio.duration }),
|
|
110
|
+
...(audio.voiceId === undefined ? {} : { voiceId: audio.voiceId }),
|
|
109
111
|
})),
|
|
110
112
|
...(entry.channelId === undefined ? {} : { channelId: entry.channelId }),
|
|
111
113
|
...(entry.channel === undefined ? {} : { channel: entry.channel }),
|
|
@@ -1,5 +1,8 @@
|
|
|
1
|
+
|
|
1
2
|
/**
|
|
2
3
|
* The AI 音频 panel: a compact audio-generation studio.
|
|
4
|
+
* TTS / music / SFX / voice design are separated; each mode only lists
|
|
5
|
+
* compatible models and shows its own parameters.
|
|
3
6
|
*/
|
|
4
7
|
|
|
5
8
|
import { useEffect, useMemo, useState } from 'react'
|
|
@@ -7,7 +10,7 @@ import type { AudiogenApi } from './api.ts'
|
|
|
7
10
|
import type { AudiogenScope } from './settings-scope.ts'
|
|
8
11
|
import { audioModelOptions } from './settings-scope.ts'
|
|
9
12
|
import { tt } from './helpers.ts'
|
|
10
|
-
import {
|
|
13
|
+
import { HISTORY_API, type AudioMode, type GeneratedAudio, type HistoryEntry } from '../protocol.ts'
|
|
11
14
|
import css from './audio-panel.module.css'
|
|
12
15
|
|
|
13
16
|
function useConfig(scope: AudiogenScope) {
|
|
@@ -47,11 +50,12 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
47
50
|
const channels = config?.channels ?? []
|
|
48
51
|
const connected = enabled && channels.some(channel => {
|
|
49
52
|
const keyHeld = scope.getSecretSetSnapshot(`channelSecrets.${channel.id}`)
|
|
50
|
-
return channel.apiUrl.trim() !== '' && keyHeld && channel.models.length > 0
|
|
53
|
+
return channel.apiUrl.trim() !== '' && keyHeld && (channel.models.length > 0 || channel.preset === 'minimax')
|
|
51
54
|
})
|
|
52
55
|
|
|
53
56
|
const [mode, setMode] = useState<AudioMode>('tts')
|
|
54
57
|
const [prompt, setPrompt] = useState('')
|
|
58
|
+
const [previewText, setPreviewText] = useState('')
|
|
55
59
|
const [model, setModel] = useState('')
|
|
56
60
|
const [voice, setVoice] = useState('')
|
|
57
61
|
const [speed, setSpeed] = useState('')
|
|
@@ -62,11 +66,18 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
62
66
|
const [outputs, setOutputs] = useState<GeneratedAudio[]>([])
|
|
63
67
|
const { entries, reload, clear } = useHistory()
|
|
64
68
|
|
|
69
|
+
const visibleModels = useMemo(() => {
|
|
70
|
+
if (mode === 'voice_design') return []
|
|
71
|
+
return modelOptions.models
|
|
72
|
+
.filter(entry => entry.category === undefined || entry.category === 'tts' && mode === 'tts' || entry.category === mode)
|
|
73
|
+
.map(entry => entry.alias)
|
|
74
|
+
}, [modelOptions.models, mode])
|
|
75
|
+
|
|
65
76
|
useEffect(() => {
|
|
66
|
-
if (
|
|
67
|
-
setModel(
|
|
77
|
+
if (visibleModels.length > 0 && !visibleModels.includes(model)) {
|
|
78
|
+
setModel(visibleModels[0]!)
|
|
68
79
|
}
|
|
69
|
-
}, [
|
|
80
|
+
}, [visibleModels, model])
|
|
70
81
|
|
|
71
82
|
const submit = async (): Promise<void> => {
|
|
72
83
|
if (prompt.trim() === '') {
|
|
@@ -78,8 +89,9 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
78
89
|
try {
|
|
79
90
|
const response = await api.generate({
|
|
80
91
|
mode,
|
|
81
|
-
model: (model ||
|
|
92
|
+
model: (model || visibleModels[0]) ?? '',
|
|
82
93
|
prompt: prompt.trim(),
|
|
94
|
+
...(previewText.trim() !== '' ? { previewText: previewText.trim() } : {}),
|
|
83
95
|
...(voice.trim() !== '' ? { voice: voice.trim() } : {}),
|
|
84
96
|
...(speed.trim() !== '' ? { speed: Number(speed) } : {}),
|
|
85
97
|
...(duration.trim() !== '' ? { duration: Number(duration) } : {}),
|
|
@@ -101,9 +113,12 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
101
113
|
const modeLabel = useMemo(() => {
|
|
102
114
|
if (mode === 'tts') return tt('mode.tts')
|
|
103
115
|
if (mode === 'music') return tt('mode.music')
|
|
104
|
-
return tt('mode.sfx')
|
|
116
|
+
if (mode === 'sfx') return tt('mode.sfx')
|
|
117
|
+
return tt('mode.voiceDesign')
|
|
105
118
|
}, [mode])
|
|
106
119
|
|
|
120
|
+
const needModel = mode !== 'voice_design'
|
|
121
|
+
|
|
107
122
|
return (
|
|
108
123
|
<div className={css.panel}>
|
|
109
124
|
<header className={css.header}>
|
|
@@ -113,7 +128,7 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
113
128
|
<div className={css.layout}>
|
|
114
129
|
<div className={css.form}>
|
|
115
130
|
<div className={css.modeRow}>
|
|
116
|
-
{(['tts', 'music', 'sfx'] as const).map(item => (
|
|
131
|
+
{(['tts', 'music', 'sfx', 'voice_design'] as const).map(item => (
|
|
117
132
|
<button
|
|
118
133
|
key={item}
|
|
119
134
|
type="button"
|
|
@@ -121,50 +136,73 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
121
136
|
data-active={mode === item ? 'true' : 'false'}
|
|
122
137
|
onClick={() => setMode(item)}
|
|
123
138
|
>
|
|
124
|
-
{item === 'tts' ? tt('mode.tts') : item === 'music' ? tt('mode.music') : tt('mode.sfx')}
|
|
139
|
+
{item === 'tts' ? tt('mode.tts') : item === 'music' ? tt('mode.music') : item === 'sfx' ? tt('mode.sfx') : tt('mode.voiceDesign')}
|
|
125
140
|
</button>
|
|
126
141
|
))}
|
|
127
142
|
</div>
|
|
143
|
+
|
|
128
144
|
<label className={css.label}>
|
|
129
|
-
<span>{mode === 'tts' ? '文本' : '提示词'}</span>
|
|
145
|
+
<span>{mode === 'voice_design' ? '音色描述' : mode === 'tts' ? '文本' : '提示词'}</span>
|
|
130
146
|
<textarea className={css.textarea} value={prompt} onChange={event => setPrompt(event.target.value)} placeholder={tt('prompt.placeholder')} />
|
|
131
147
|
</label>
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
<
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
</
|
|
138
|
-
|
|
148
|
+
|
|
149
|
+
{mode === 'voice_design' ? (
|
|
150
|
+
<label className={css.label}>
|
|
151
|
+
<span>试听文本</span>
|
|
152
|
+
<input className={css.input} value={previewText} onChange={event => setPreviewText(event.target.value)} placeholder="你好,这是新设计的音色试听。" />
|
|
153
|
+
</label>
|
|
154
|
+
) : null}
|
|
155
|
+
|
|
156
|
+
{needModel ? (
|
|
157
|
+
<label className={css.label}>
|
|
158
|
+
<span>{tt('model.label')}</span>
|
|
159
|
+
<select className={css.select} value={model} onChange={event => setModel(event.target.value)}>
|
|
160
|
+
{visibleModels.length === 0 ? <option value="">(当前模式暂无可用模型)</option> : null}
|
|
161
|
+
{visibleModels.map(item => <option key={item} value={item}>{item}</option>)}
|
|
162
|
+
</select>
|
|
163
|
+
</label>
|
|
164
|
+
) : null}
|
|
165
|
+
|
|
139
166
|
{mode === 'tts' ? (
|
|
140
167
|
<label className={css.label}>
|
|
141
168
|
<span>{tt('voice.label')}</span>
|
|
142
169
|
<input className={css.input} value={voice} onChange={event => setVoice(event.target.value)} placeholder="alloy / 自定义音色" />
|
|
143
170
|
</label>
|
|
144
171
|
) : null}
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
<
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
<
|
|
155
|
-
|
|
156
|
-
<
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
172
|
+
|
|
173
|
+
{mode === 'tts' ? (
|
|
174
|
+
<label className={css.label}>
|
|
175
|
+
<span>{tt('speed.label')}</span>
|
|
176
|
+
<input className={css.input} type="number" step="0.1" min="0.5" max="2" value={speed} onChange={event => setSpeed(event.target.value)} placeholder="1.0" />
|
|
177
|
+
</label>
|
|
178
|
+
) : null}
|
|
179
|
+
|
|
180
|
+
{mode === 'music' || mode === 'sfx' ? (
|
|
181
|
+
<label className={css.label}>
|
|
182
|
+
<span>{tt('duration.label')}</span>
|
|
183
|
+
<input className={css.input} type="number" step="1" min="1" max="120" value={duration} onChange={event => setDuration(event.target.value)} placeholder="30" />
|
|
184
|
+
</label>
|
|
185
|
+
) : null}
|
|
186
|
+
|
|
187
|
+
{needModel ? (
|
|
188
|
+
<label className={css.label}>
|
|
189
|
+
<span>{tt('format.label')}</span>
|
|
190
|
+
<select className={css.select} value={format} onChange={event => setFormat(event.target.value)}>
|
|
191
|
+
<option value="mp3">mp3</option>
|
|
192
|
+
<option value="wav">wav</option>
|
|
193
|
+
<option value="flac">flac</option>
|
|
194
|
+
<option value="ogg">ogg</option>
|
|
195
|
+
<option value="pcm">pcm</option>
|
|
196
|
+
</select>
|
|
197
|
+
</label>
|
|
198
|
+
) : null}
|
|
199
|
+
|
|
163
200
|
{!connected && <p className={css.hint}>{tt('config.missing')}</p>}
|
|
164
|
-
<button type="button" className={css.generate} disabled={loading || !connected} onClick={() => void submit()}>
|
|
201
|
+
<button type="button" className={css.generate} disabled={loading || !connected || (needModel && visibleModels.length === 0)} onClick={() => void submit()}>
|
|
165
202
|
{loading ? tt('generating') : tt('generate')}
|
|
166
203
|
</button>
|
|
167
204
|
</div>
|
|
205
|
+
|
|
168
206
|
<div className={css.result}>
|
|
169
207
|
{error !== null ? <p className={css.error}>{error}</p> : null}
|
|
170
208
|
{outputs.length === 0 ? <p className={css.empty}>{tt('result.empty')}</p> : (
|
|
@@ -173,6 +211,7 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
173
211
|
<div className={css.audioList}>
|
|
174
212
|
{outputs.map((audio, index) => (
|
|
175
213
|
<div className={css.audioCard} key={audio.id}>
|
|
214
|
+
{audio.voiceId !== undefined ? <p className={css.hint}>新音色 ID:{audio.voiceId}</p> : null}
|
|
176
215
|
<audio className={css.audio} controls preload="metadata" src={dataUrlOf(audio)} />
|
|
177
216
|
<a className={css.download} href={dataUrlOf(audio)} download={`generated-${index + 1}.${audio.mime.split('/')[1]?.replace('mpeg', 'mp3') ?? 'mp3'}`}>下载</a>
|
|
178
217
|
</div>
|
|
@@ -181,6 +220,7 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
|
|
|
181
220
|
</>
|
|
182
221
|
)}
|
|
183
222
|
</div>
|
|
223
|
+
|
|
184
224
|
<aside className={css.history}>
|
|
185
225
|
<div className={css.historyHeader}>
|
|
186
226
|
<strong className={css.historyTitle}>{tt('history.title')}</strong>
|