dsh-audiogen 0.4.0 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +665 -407
- package/lib/client.js.map +1 -1
- package/lib/index.js +289 -157
- package/package.json +1 -1
- package/skills/design/SKILL.md +5 -0
- package/skills/music/SKILL.md +5 -0
- package/skills/sfx/SKILL.md +5 -0
- package/skills/tts/SKILL.md +5 -0
- package/src/agent-audio-tools.ts +255 -149
- package/src/client/audio-panel.module.css +87 -0
- package/src/client/studio-view.tsx +270 -91
- package/src/index.ts +32 -0
- package/src/protocol.ts +1 -1
package/src/agent-audio-tools.ts
CHANGED
|
@@ -33,6 +33,14 @@ interface SavedAudioRef extends AgentAudioRef {
|
|
|
33
33
|
file: string
|
|
34
34
|
}
|
|
35
35
|
|
|
36
|
+
/** Per-model group in a multi-model comparison result. */
|
|
37
|
+
interface AgentAudioGroup {
|
|
38
|
+
model: string
|
|
39
|
+
audio: AgentAudioRef[]
|
|
40
|
+
resources?: string[]
|
|
41
|
+
error?: string
|
|
42
|
+
}
|
|
43
|
+
|
|
36
44
|
interface AgentAudioResult {
|
|
37
45
|
status: string
|
|
38
46
|
message: string
|
|
@@ -41,6 +49,8 @@ interface AgentAudioResult {
|
|
|
41
49
|
audio: AgentAudioRef[]
|
|
42
50
|
/** Resource-library entry ids when the audio was saved to the library. */
|
|
43
51
|
resources?: string[]
|
|
52
|
+
/** Per-model results when several models were generated with the same prompt. */
|
|
53
|
+
groups?: AgentAudioGroup[]
|
|
44
54
|
error?: string
|
|
45
55
|
}
|
|
46
56
|
|
|
@@ -56,6 +66,17 @@ const audioRefSchema = {
|
|
|
56
66
|
},
|
|
57
67
|
} as const
|
|
58
68
|
|
|
69
|
+
const groupSchema = {
|
|
70
|
+
type: 'object',
|
|
71
|
+
additionalProperties: false,
|
|
72
|
+
properties: {
|
|
73
|
+
model: { type: 'string', required: true },
|
|
74
|
+
audio: { type: 'array', required: true, items: audioRefSchema },
|
|
75
|
+
resources: { type: 'array', items: { type: 'string' } },
|
|
76
|
+
error: { type: 'string' },
|
|
77
|
+
},
|
|
78
|
+
} as const
|
|
79
|
+
|
|
59
80
|
const resultSchema = {
|
|
60
81
|
type: 'object',
|
|
61
82
|
additionalProperties: false,
|
|
@@ -66,6 +87,7 @@ const resultSchema = {
|
|
|
66
87
|
model: { type: 'string', required: true },
|
|
67
88
|
audio: { type: 'array', required: true, items: audioRefSchema },
|
|
68
89
|
resources: { type: 'array', items: { type: 'string' } },
|
|
90
|
+
groups: { type: 'array', items: groupSchema },
|
|
69
91
|
error: { type: 'string' },
|
|
70
92
|
},
|
|
71
93
|
} as const
|
|
@@ -120,6 +142,16 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
120
142
|
prompt: { type: 'string', required: true, description: 'For tts, the text to speak. For music/sfx, a descriptive prompt.' },
|
|
121
143
|
mode: { type: 'string', enum: ['tts', 'music', 'sfx', 'voice_design'], description: 'Generation mode. Defaults to tts.' },
|
|
122
144
|
model: { type: 'string', description: 'One of the configured audio models/voices. Defaults to the first configured model.' },
|
|
145
|
+
models: {
|
|
146
|
+
type: 'array',
|
|
147
|
+
items: { type: 'string' },
|
|
148
|
+
description: 'Optional: several configured model aliases to generate the SAME prompt with each one, sequentially, for comparison (e.g. ["speech-2.8-hd","speech-2.6-hd"]). Cannot be combined with model; when present, models wins.',
|
|
149
|
+
},
|
|
150
|
+
model_params: {
|
|
151
|
+
type: 'object',
|
|
152
|
+
additionalProperties: true,
|
|
153
|
+
description: 'Optional per-model parameter overrides used with "models" (automatic by default = all models share the global params). Keys are model aliases; values are partial param objects using the same param names (format, duration, voice, speed, emotion, vol, pitch, sample_rate, bitrate, lyrics, is_instrumental, loop, prompt_influence, seed, steps, cfg_scale, subtitle_enable, aigc_watermark, language_boost, pronunciation_tone, voice_modify, timbre_weights). Unset fields fall back to the global values.',
|
|
154
|
+
},
|
|
123
155
|
voice: { type: 'string', description: 'Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio.' },
|
|
124
156
|
preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
|
|
125
157
|
speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
|
|
@@ -163,10 +195,9 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
163
195
|
type: 'object',
|
|
164
196
|
additionalProperties: false,
|
|
165
197
|
properties: {
|
|
166
|
-
voice_id: { type: 'string' },
|
|
167
|
-
weight: { type: 'integer' },
|
|
198
|
+
voice_id: { type: 'string', required: true },
|
|
199
|
+
weight: { type: 'integer', required: true },
|
|
168
200
|
},
|
|
169
|
-
required: ['voice_id', 'weight'],
|
|
170
201
|
},
|
|
171
202
|
description: 'MiniMax TTS dual-voice blend weights (timbre_weights).',
|
|
172
203
|
},
|
|
@@ -185,170 +216,245 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
185
216
|
async execute(args, exec) {
|
|
186
217
|
const config = resolve()
|
|
187
218
|
ensureConfigured(config)
|
|
188
|
-
const mode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
: {}),
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
...(typeof args.language_boost === 'string' && args.language_boost.trim() !== '' ? { languageBoost: args.language_boost.trim() } : {}),
|
|
249
|
-
...(voiceModify !== undefined ? { voiceModify } : {}),
|
|
250
|
-
...(timbreWeights !== undefined && timbreWeights.length > 0 ? { timbreWeights } : {}),
|
|
219
|
+
const mode: AudioMode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
|
|
220
|
+
/** 把生成参数(snake_case 入参或 model_params 片段)映射为请求字段。 */
|
|
221
|
+
const mapParams = (raw: Record<string, unknown>): Partial<GenerateAudioRequest> => {
|
|
222
|
+
const voiceModify = typeof raw.voice_modify === 'object' && raw.voice_modify !== null
|
|
223
|
+
? (() => {
|
|
224
|
+
const src = raw.voice_modify as Record<string, unknown>
|
|
225
|
+
const out: { pitch?: number; intensity?: number; timbre?: number; soundEffects?: string } = {}
|
|
226
|
+
if (typeof src.pitch === 'number') out.pitch = src.pitch
|
|
227
|
+
if (typeof src.intensity === 'number') out.intensity = src.intensity
|
|
228
|
+
if (typeof src.timbre === 'number') out.timbre = src.timbre
|
|
229
|
+
if (typeof src.sound_effects === 'string' && src.sound_effects.trim() !== '') out.soundEffects = src.sound_effects.trim()
|
|
230
|
+
return Object.keys(out).length > 0 ? out : undefined
|
|
231
|
+
})()
|
|
232
|
+
: undefined
|
|
233
|
+
const timbreWeights = Array.isArray(raw.timbre_weights)
|
|
234
|
+
? raw.timbre_weights
|
|
235
|
+
.filter((item): item is { voice_id: string; weight: number } => typeof item === 'object' && item !== null && typeof (item as { voice_id?: unknown }).voice_id === 'string' && typeof (item as { weight?: unknown }).weight === 'number')
|
|
236
|
+
.map(item => ({ voiceId: (item.voice_id as string).trim(), weight: item.weight as number }))
|
|
237
|
+
.filter(item => item.voiceId !== '')
|
|
238
|
+
: undefined
|
|
239
|
+
const stringOrEmpty = (key: string): string | undefined => {
|
|
240
|
+
const value = raw[key]
|
|
241
|
+
return typeof value === 'string' && value.trim() !== '' ? value.trim() : undefined
|
|
242
|
+
}
|
|
243
|
+
const finiteOrUndefined = (key: string): number | undefined => {
|
|
244
|
+
const value = raw[key]
|
|
245
|
+
return typeof value === 'number' && Number.isFinite(value) ? value : undefined
|
|
246
|
+
}
|
|
247
|
+
return {
|
|
248
|
+
...(stringOrEmpty('voice') !== undefined ? { voice: stringOrEmpty('voice')! } : {}),
|
|
249
|
+
...(stringOrEmpty('preview_text') !== undefined ? { previewText: stringOrEmpty('preview_text')! } : {}),
|
|
250
|
+
...(finiteOrUndefined('speed') !== undefined ? { speed: finiteOrUndefined('speed')! } : {}),
|
|
251
|
+
...(finiteOrUndefined('duration') !== undefined ? { duration: finiteOrUndefined('duration')! } : {}),
|
|
252
|
+
...(stringOrEmpty('lyrics') !== undefined ? { lyrics: stringOrEmpty('lyrics')! } : {}),
|
|
253
|
+
...(typeof raw.is_instrumental === 'boolean' ? { isInstrumental: raw.is_instrumental } : {}),
|
|
254
|
+
...(typeof raw.loop === 'boolean' ? { loop: raw.loop } : {}),
|
|
255
|
+
...(finiteOrUndefined('prompt_influence') !== undefined ? { promptInfluence: finiteOrUndefined('prompt_influence')! } : {}),
|
|
256
|
+
...(finiteOrUndefined('seed') !== undefined ? { seed: finiteOrUndefined('seed')! } : {}),
|
|
257
|
+
...(finiteOrUndefined('steps') !== undefined ? { steps: finiteOrUndefined('steps')! } : {}),
|
|
258
|
+
...(finiteOrUndefined('cfg_scale') !== undefined ? { cfgScale: finiteOrUndefined('cfg_scale')! } : {}),
|
|
259
|
+
...(stringOrEmpty('format') !== undefined ? { format: stringOrEmpty('format')! } : {}),
|
|
260
|
+
// ---- MiniMax / ElevenLabs / Stability 专属字段 ----
|
|
261
|
+
...(stringOrEmpty('emotion') !== undefined ? { emotion: stringOrEmpty('emotion')! } : {}),
|
|
262
|
+
...(finiteOrUndefined('vol') !== undefined ? { vol: finiteOrUndefined('vol')! } : {}),
|
|
263
|
+
...(finiteOrUndefined('pitch') !== undefined ? { pitch: finiteOrUndefined('pitch')! } : {}),
|
|
264
|
+
...(typeof raw.text_normalization === 'boolean' ? { textNormalization: raw.text_normalization } : {}),
|
|
265
|
+
...(typeof raw.latex_read === 'boolean' ? { latexRead: raw.latex_read } : {}),
|
|
266
|
+
...(Array.isArray(raw.pronunciation_tone) && raw.pronunciation_tone.length > 0
|
|
267
|
+
? { pronunciationTone: raw.pronunciation_tone.filter((item): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()) }
|
|
268
|
+
: {}),
|
|
269
|
+
...(finiteOrUndefined('sample_rate') !== undefined ? { sampleRate: finiteOrUndefined('sample_rate')! } : {}),
|
|
270
|
+
...(finiteOrUndefined('bitrate') !== undefined ? { bitrate: finiteOrUndefined('bitrate')! } : {}),
|
|
271
|
+
...(finiteOrUndefined('channel') !== undefined ? { audioChannel: finiteOrUndefined('channel')! } : {}),
|
|
272
|
+
...(typeof raw.force_cbr === 'boolean' ? { forceCbr: raw.force_cbr } : {}),
|
|
273
|
+
...(typeof raw.subtitle_enable === 'boolean' ? { subtitleEnable: raw.subtitle_enable } : {}),
|
|
274
|
+
...(typeof raw.aigc_watermark === 'boolean' ? { aigcWatermark: raw.aigc_watermark } : {}),
|
|
275
|
+
...(stringOrEmpty('language_boost') !== undefined ? { languageBoost: stringOrEmpty('language_boost')! } : {}),
|
|
276
|
+
...(voiceModify !== undefined ? { voiceModify } : {}),
|
|
277
|
+
...(timbreWeights !== undefined && timbreWeights.length > 0 ? { timbreWeights } : {}),
|
|
278
|
+
}
|
|
251
279
|
}
|
|
252
|
-
|
|
253
|
-
const
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
const
|
|
258
|
-
|
|
259
|
-
id: stored.id,
|
|
260
|
-
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
261
|
-
file: stored.file,
|
|
262
|
-
mime: stored.mime,
|
|
263
|
-
bytes: stored.bytes,
|
|
264
|
-
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
265
|
-
})
|
|
266
|
-
audio.push({
|
|
267
|
-
id: stored.id,
|
|
268
|
-
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
269
|
-
mime: stored.mime,
|
|
270
|
-
bytes: stored.bytes,
|
|
271
|
-
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
272
|
-
})
|
|
280
|
+
const buildRequest = (picked: { channel: AudioChannel; alias: string; upstream: string }): GenerateAudioRequest => {
|
|
281
|
+
const base = mapParams(args as unknown as Record<string, unknown>)
|
|
282
|
+
// 每模型参数覆盖(model_params[alias]);缺省 = 自动沿用全局配置
|
|
283
|
+
let override: Partial<GenerateAudioRequest> = {}
|
|
284
|
+
if (typeof args.model_params === 'object' && args.model_params !== null) {
|
|
285
|
+
const perModel = (args.model_params as Record<string, unknown>)[picked.alias]
|
|
286
|
+
if (typeof perModel === 'object' && perModel !== null) override = mapParams(perModel as Record<string, unknown>)
|
|
273
287
|
}
|
|
288
|
+
return {
|
|
289
|
+
mode,
|
|
290
|
+
model: picked.alias,
|
|
291
|
+
upstream: picked.upstream,
|
|
292
|
+
channelId: picked.channel.id,
|
|
293
|
+
channel: picked.channel.name,
|
|
294
|
+
prompt: typeof args.prompt === 'string' ? args.prompt.trim() : '',
|
|
295
|
+
...base,
|
|
296
|
+
...override,
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
/** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
|
|
300
|
+
const runOne = async (picked: { channel: AudioChannel; alias: string; upstream: string }): Promise<AgentAudioGroup> => {
|
|
301
|
+
const request = buildRequest(picked)
|
|
274
302
|
try {
|
|
275
|
-
await
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
id: saved[index]!.id,
|
|
287
|
-
file: saved[index]!.file,
|
|
288
|
-
b64: Buffer.from(output.data).toString('base64'),
|
|
289
|
-
mime: saved[index]!.mime,
|
|
290
|
-
bytes: saved[index]!.bytes,
|
|
291
|
-
url: saved[index]!.url,
|
|
303
|
+
const outputs = await generateAudio(picked.channel, request, exec.signal)
|
|
304
|
+
const audio: AgentAudioRef[] = []
|
|
305
|
+
const saved: SavedAudioRef[] = []
|
|
306
|
+
for (const [index, output] of outputs.entries()) {
|
|
307
|
+
const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`)
|
|
308
|
+
saved.push({
|
|
309
|
+
id: stored.id,
|
|
310
|
+
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
311
|
+
file: stored.file,
|
|
312
|
+
mime: stored.mime,
|
|
313
|
+
bytes: stored.bytes,
|
|
292
314
|
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
293
|
-
})
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
const wantSave = args.save_to_library === true || (config.autoSaveToLibrary && args.save_to_library !== false)
|
|
303
|
-
let resources: string[] | undefined
|
|
304
|
-
if (wantSave) {
|
|
315
|
+
})
|
|
316
|
+
audio.push({
|
|
317
|
+
id: stored.id,
|
|
318
|
+
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
319
|
+
mime: stored.mime,
|
|
320
|
+
bytes: stored.bytes,
|
|
321
|
+
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
322
|
+
})
|
|
323
|
+
}
|
|
305
324
|
try {
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
325
|
+
await appendHistory({
|
|
326
|
+
id: randomUUID(),
|
|
327
|
+
createdAt: Date.now(),
|
|
328
|
+
mode: request.mode,
|
|
329
|
+
model: picked.alias,
|
|
330
|
+
prompt: request.prompt,
|
|
331
|
+
...(request.voice === undefined ? {} : { voice: request.voice }),
|
|
332
|
+
...(request.speed === undefined ? {} : { speed: request.speed }),
|
|
333
|
+
...(request.duration === undefined ? {} : { duration: request.duration }),
|
|
334
|
+
...(request.format === undefined ? {} : { format: request.format }),
|
|
335
|
+
audio: outputs.map((output, index) => ({
|
|
336
|
+
id: saved[index]!.id,
|
|
337
|
+
file: saved[index]!.file,
|
|
338
|
+
b64: Buffer.from(output.data).toString('base64'),
|
|
339
|
+
mime: saved[index]!.mime,
|
|
340
|
+
bytes: saved[index]!.bytes,
|
|
341
|
+
url: saved[index]!.url,
|
|
342
|
+
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
312
343
|
})),
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
provenance: {
|
|
317
|
-
mode: request.mode,
|
|
318
|
-
prompt: request.prompt,
|
|
319
|
-
channel: picked.channel.name,
|
|
320
|
-
channelId: picked.channel.id,
|
|
321
|
-
apiUrl: picked.channel.apiUrl,
|
|
322
|
-
model: picked.alias,
|
|
323
|
-
upstream: picked.upstream,
|
|
324
|
-
...(request.voice === undefined ? {} : { voice: request.voice }),
|
|
325
|
-
params: { ...request },
|
|
326
|
-
},
|
|
344
|
+
channelId: picked.channel.id,
|
|
345
|
+
channel: picked.channel.name,
|
|
346
|
+
params: { ...request },
|
|
327
347
|
})
|
|
328
|
-
resources = [entry.id]
|
|
329
348
|
} catch {
|
|
330
|
-
//
|
|
349
|
+
// History is best-effort and must not fail the agent tool.
|
|
350
|
+
}
|
|
351
|
+
// ---- 资源库保存:显式参数优先;设置自动入库时可用 false 跳过 ----
|
|
352
|
+
const wantSave = args.save_to_library === true || (config.autoSaveToLibrary && args.save_to_library !== false)
|
|
353
|
+
let resources: string[] | undefined
|
|
354
|
+
if (wantSave) {
|
|
355
|
+
try {
|
|
356
|
+
const entry = await saveToLibrary({
|
|
357
|
+
audioFiles: saved.map(item => ({
|
|
358
|
+
id: item.id,
|
|
359
|
+
file: item.file,
|
|
360
|
+
mime: item.mime,
|
|
361
|
+
...(item.voiceId === undefined ? {} : { voiceId: item.voiceId }),
|
|
362
|
+
})),
|
|
363
|
+
type: libraryTypeOf(request.mode, args.library_type),
|
|
364
|
+
...(typeof args.library_name === 'string' && args.library_name.trim() !== '' ? { name: args.library_name.trim() } : {}),
|
|
365
|
+
...(Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag): tag is string => typeof tag === 'string' && tag.trim() !== '').map(tag => tag.trim()) } : {}),
|
|
366
|
+
provenance: {
|
|
367
|
+
mode: request.mode,
|
|
368
|
+
prompt: request.prompt,
|
|
369
|
+
channel: picked.channel.name,
|
|
370
|
+
channelId: picked.channel.id,
|
|
371
|
+
apiUrl: picked.channel.apiUrl,
|
|
372
|
+
model: picked.alias,
|
|
373
|
+
upstream: picked.upstream,
|
|
374
|
+
...(request.voice === undefined ? {} : { voice: request.voice }),
|
|
375
|
+
params: { ...request },
|
|
376
|
+
},
|
|
377
|
+
})
|
|
378
|
+
resources = [entry.id]
|
|
379
|
+
} catch {
|
|
380
|
+
// library-save is best-effort; generation already succeeded.
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
return {
|
|
384
|
+
model: picked.alias,
|
|
385
|
+
audio,
|
|
386
|
+
...(resources === undefined ? {} : { resources }),
|
|
387
|
+
}
|
|
388
|
+
} catch (error) {
|
|
389
|
+
if (exec.signal?.aborted === true) throw error
|
|
390
|
+
return {
|
|
391
|
+
model: picked.alias,
|
|
392
|
+
audio: [],
|
|
393
|
+
error: error instanceof Error ? error.message : String(error),
|
|
331
394
|
}
|
|
332
395
|
}
|
|
396
|
+
}
|
|
397
|
+
// ---- 多模型对比:同一 prompt 依次生成 ----
|
|
398
|
+
const requestedModels = Array.isArray(args.models)
|
|
399
|
+
? [...new Set(args.models.filter((item: unknown): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()))]
|
|
400
|
+
: []
|
|
401
|
+
if (requestedModels.length > 0 && mode !== 'voice_design') {
|
|
402
|
+
const groups: AgentAudioGroup[] = []
|
|
403
|
+
let succeeded = 0
|
|
404
|
+
for (const alias of requestedModels) {
|
|
405
|
+
let picked
|
|
406
|
+
try {
|
|
407
|
+
picked = resolveModel(config, alias)
|
|
408
|
+
} catch (error) {
|
|
409
|
+
groups.push({ model: alias, audio: [], error: error instanceof Error ? error.message : String(error) })
|
|
410
|
+
continue
|
|
411
|
+
}
|
|
412
|
+
const group = await runOne(picked)
|
|
413
|
+
groups.push(group)
|
|
414
|
+
if (group.error === undefined) succeeded++
|
|
415
|
+
}
|
|
333
416
|
return {
|
|
334
|
-
status: 'completed',
|
|
335
|
-
message:
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
417
|
+
status: succeeded > 0 ? 'completed' : 'failed',
|
|
418
|
+
message: succeeded > 0
|
|
419
|
+
? `Generated ${succeeded}/${groups.length} model(s) with the same prompt for comparison. The audio files can be played/downloaded from the returned URLs.`
|
|
420
|
+
: 'All model generations failed.',
|
|
421
|
+
mode,
|
|
422
|
+
model: groups[0]?.model ?? requestedModels[0]!,
|
|
423
|
+
audio: groups.flatMap(group => group.audio),
|
|
424
|
+
groups,
|
|
425
|
+
...(succeeded === 0
|
|
426
|
+
? { error: groups.map(group => `${group.model}: ${group.error ?? ''}`).filter(item => !item.endsWith(': ')).join(';') }
|
|
427
|
+
: {}),
|
|
340
428
|
}
|
|
341
|
-
}
|
|
342
|
-
|
|
429
|
+
}
|
|
430
|
+
// ---- 单模型(含音色设计) ----
|
|
431
|
+
const picked = mode === 'voice_design'
|
|
432
|
+
? (() => {
|
|
433
|
+
const usable = config.channels.filter(channel => channel.apiUrl.trim() !== '' && channel.apiKey.trim() !== '')
|
|
434
|
+
const target = usable.find(channel => channel.id === config.defaultChannelId) ?? usable[0]
|
|
435
|
+
if (target === undefined) throw new AudioGenError('No usable audio channel is configured for voice design.', 'no-channel-available')
|
|
436
|
+
return { channel: target, alias: '', upstream: '' }
|
|
437
|
+
})()
|
|
438
|
+
: resolveModel(config, args.model)
|
|
439
|
+
const one = await runOne(picked)
|
|
440
|
+
if (one.error !== undefined) {
|
|
343
441
|
return {
|
|
344
442
|
status: 'failed',
|
|
345
443
|
message: 'Audio generation failed.',
|
|
346
|
-
mode
|
|
347
|
-
model:
|
|
444
|
+
mode,
|
|
445
|
+
model: one.model,
|
|
348
446
|
audio: [],
|
|
349
|
-
error:
|
|
447
|
+
error: one.error,
|
|
350
448
|
}
|
|
351
449
|
}
|
|
450
|
+
return {
|
|
451
|
+
status: 'completed',
|
|
452
|
+
message: 'Audio generation completed. The audio files can be played/downloaded from the returned URLs.',
|
|
453
|
+
mode,
|
|
454
|
+
model: one.model,
|
|
455
|
+
audio: one.audio,
|
|
456
|
+
...(one.resources === undefined ? {} : { resources: one.resources }),
|
|
457
|
+
}
|
|
352
458
|
},
|
|
353
459
|
}))
|
|
354
460
|
|
|
@@ -805,3 +805,90 @@
|
|
|
805
805
|
@media (prefers-reduced-motion: reduce) {
|
|
806
806
|
.toast { animation-duration: 1ms; }
|
|
807
807
|
}
|
|
808
|
+
|
|
809
|
+
/* 多模型对比 */
|
|
810
|
+
.modelCheckList {
|
|
811
|
+
max-height: 180px;
|
|
812
|
+
overflow-y: auto;
|
|
813
|
+
display: flex;
|
|
814
|
+
flex-direction: column;
|
|
815
|
+
gap: 4px;
|
|
816
|
+
padding: 8px;
|
|
817
|
+
border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
|
|
818
|
+
border-radius: 8px;
|
|
819
|
+
background: var(--dsw-alias-bg-layer-3, #fff);
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
.resultGroups {
|
|
823
|
+
display: flex;
|
|
824
|
+
flex-direction: column;
|
|
825
|
+
gap: 16px;
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
.resultGroup {
|
|
829
|
+
display: flex;
|
|
830
|
+
flex-direction: column;
|
|
831
|
+
gap: 8px;
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
.resultGroupHead {
|
|
835
|
+
display: flex;
|
|
836
|
+
align-items: center;
|
|
837
|
+
gap: 8px;
|
|
838
|
+
flex-wrap: wrap;
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
.resultGroupChip {
|
|
842
|
+
font-size: 12px;
|
|
843
|
+
font-weight: 600;
|
|
844
|
+
padding: 3px 10px;
|
|
845
|
+
border-radius: 999px;
|
|
846
|
+
background: var(--dsw-alias-bg-hover, #f3f4f6);
|
|
847
|
+
border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
|
|
848
|
+
color: var(--dsw-alias-label-primary, #1f2328);
|
|
849
|
+
}
|
|
850
|
+
|
|
851
|
+
.resultGroupError {
|
|
852
|
+
font-size: 12px;
|
|
853
|
+
color: #dc2626;
|
|
854
|
+
}
|
|
855
|
+
|
|
856
|
+
.resultGroupCount {
|
|
857
|
+
font-size: 11px;
|
|
858
|
+
color: var(--dsw-alias-label-tertiary, #9ca3af);
|
|
859
|
+
}
|
|
860
|
+
|
|
861
|
+
/* 每模型参数覆盖矩阵 */
|
|
862
|
+
.overrideTable {
|
|
863
|
+
display: flex;
|
|
864
|
+
flex-direction: column;
|
|
865
|
+
gap: 6px;
|
|
866
|
+
}
|
|
867
|
+
|
|
868
|
+
.overrideRow {
|
|
869
|
+
display: flex;
|
|
870
|
+
gap: 6px;
|
|
871
|
+
align-items: center;
|
|
872
|
+
}
|
|
873
|
+
|
|
874
|
+
.overrideCell {
|
|
875
|
+
flex: 1;
|
|
876
|
+
min-width: 0;
|
|
877
|
+
font-size: 12px;
|
|
878
|
+
font-weight: 500;
|
|
879
|
+
color: var(--dsw-alias-label-secondary, #6b7280);
|
|
880
|
+
overflow: hidden;
|
|
881
|
+
text-overflow: ellipsis;
|
|
882
|
+
white-space: nowrap;
|
|
883
|
+
}
|
|
884
|
+
|
|
885
|
+
.overrideCell .input {
|
|
886
|
+
min-height: 30px;
|
|
887
|
+
padding: 5px 8px;
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
.overrideCellHead {
|
|
891
|
+
font-weight: 700;
|
|
892
|
+
color: var(--dsw-alias-label-primary, #1f2328);
|
|
893
|
+
white-space: normal;
|
|
894
|
+
}
|