dsh-audiogen 0.4.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -33,6 +33,14 @@ interface SavedAudioRef extends AgentAudioRef {
33
33
  file: string
34
34
  }
35
35
 
36
+ /** Per-model group in a multi-model comparison result. */
37
+ interface AgentAudioGroup {
38
+ model: string
39
+ audio: AgentAudioRef[]
40
+ resources?: string[]
41
+ error?: string
42
+ }
43
+
36
44
  interface AgentAudioResult {
37
45
  status: string
38
46
  message: string
@@ -41,6 +49,8 @@ interface AgentAudioResult {
41
49
  audio: AgentAudioRef[]
42
50
  /** Resource-library entry ids when the audio was saved to the library. */
43
51
  resources?: string[]
52
+ /** Per-model results when several models were generated with the same prompt. */
53
+ groups?: AgentAudioGroup[]
44
54
  error?: string
45
55
  }
46
56
 
@@ -56,6 +66,17 @@ const audioRefSchema = {
56
66
  },
57
67
  } as const
58
68
 
69
+ const groupSchema = {
70
+ type: 'object',
71
+ additionalProperties: false,
72
+ properties: {
73
+ model: { type: 'string', required: true },
74
+ audio: { type: 'array', required: true, items: audioRefSchema },
75
+ resources: { type: 'array', items: { type: 'string' } },
76
+ error: { type: 'string' },
77
+ },
78
+ } as const
79
+
59
80
  const resultSchema = {
60
81
  type: 'object',
61
82
  additionalProperties: false,
@@ -66,6 +87,7 @@ const resultSchema = {
66
87
  model: { type: 'string', required: true },
67
88
  audio: { type: 'array', required: true, items: audioRefSchema },
68
89
  resources: { type: 'array', items: { type: 'string' } },
90
+ groups: { type: 'array', items: groupSchema },
69
91
  error: { type: 'string' },
70
92
  },
71
93
  } as const
@@ -120,6 +142,16 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
120
142
  prompt: { type: 'string', required: true, description: 'For tts, the text to speak. For music/sfx, a descriptive prompt.' },
121
143
  mode: { type: 'string', enum: ['tts', 'music', 'sfx', 'voice_design'], description: 'Generation mode. Defaults to tts.' },
122
144
  model: { type: 'string', description: 'One of the configured audio models/voices. Defaults to the first configured model.' },
145
+ models: {
146
+ type: 'array',
147
+ items: { type: 'string' },
148
+ description: 'Optional: several configured model aliases to generate the SAME prompt with each one, sequentially, for comparison (e.g. ["speech-2.8-hd","speech-2.6-hd"]). Cannot be combined with model; when present, models wins.',
149
+ },
150
+ model_params: {
151
+ type: 'object',
152
+ additionalProperties: true,
153
+ description: 'Optional per-model parameter overrides used with "models" (automatic by default = all models share the global params). Keys are model aliases; values are partial param objects using the same param names (format, duration, voice, speed, emotion, vol, pitch, sample_rate, bitrate, lyrics, is_instrumental, loop, prompt_influence, seed, steps, cfg_scale, subtitle_enable, aigc_watermark, language_boost, pronunciation_tone, voice_modify, timbre_weights). Unset fields fall back to the global values.',
154
+ },
123
155
  voice: { type: 'string', description: 'Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio.' },
124
156
  preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
125
157
  speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
@@ -163,10 +195,9 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
163
195
  type: 'object',
164
196
  additionalProperties: false,
165
197
  properties: {
166
- voice_id: { type: 'string' },
167
- weight: { type: 'integer' },
198
+ voice_id: { type: 'string', required: true },
199
+ weight: { type: 'integer', required: true },
168
200
  },
169
- required: ['voice_id', 'weight'],
170
201
  },
171
202
  description: 'MiniMax TTS dual-voice blend weights (timbre_weights).',
172
203
  },
@@ -185,170 +216,245 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
185
216
  async execute(args, exec) {
186
217
  const config = resolve()
187
218
  ensureConfigured(config)
188
- const mode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
189
- const picked = mode === 'voice_design'
190
- ? (() => {
191
- const usable = config.channels.filter(channel => channel.apiUrl.trim() !== '' && channel.apiKey.trim() !== '')
192
- const target = usable.find(channel => channel.id === config.defaultChannelId) ?? usable[0]
193
- if (target === undefined) throw new AudioGenError('No usable audio channel is configured for voice design.', 'no-channel-available')
194
- return { channel: target, alias: '', upstream: '' }
195
- })()
196
- : resolveModel(config, args.model)
197
- const voiceModify = typeof args.voice_modify === 'object' && args.voice_modify !== null
198
- ? (() => {
199
- const raw = args.voice_modify as Record<string, unknown>
200
- const out: { pitch?: number; intensity?: number; timbre?: number; soundEffects?: string } = {}
201
- if (typeof raw.pitch === 'number') out.pitch = raw.pitch
202
- if (typeof raw.intensity === 'number') out.intensity = raw.intensity
203
- if (typeof raw.timbre === 'number') out.timbre = raw.timbre
204
- if (typeof raw.sound_effects === 'string' && raw.sound_effects.trim() !== '') out.soundEffects = raw.sound_effects.trim()
205
- return Object.keys(out).length > 0 ? out : undefined
206
- })()
207
- : undefined
208
- const timbreWeights = Array.isArray(args.timbre_weights)
209
- ? args.timbre_weights
210
- .filter((item): item is { voice_id: string; weight: number } => typeof item === 'object' && item !== null && typeof (item as { voice_id?: unknown }).voice_id === 'string' && typeof (item as { weight?: unknown }).weight === 'number')
211
- .map(item => ({ voiceId: (item.voice_id as string).trim(), weight: item.weight as number }))
212
- .filter(item => item.voiceId !== '')
213
- : undefined
214
- const request: GenerateAudioRequest = {
215
- mode,
216
- model: picked.alias,
217
- upstream: picked.upstream,
218
- channelId: picked.channel.id,
219
- channel: picked.channel.name,
220
- prompt: args.prompt.trim(),
221
- ...(typeof args.voice === 'string' && args.voice.trim() !== '' ? { voice: args.voice.trim() } : {}),
222
- ...(typeof args.preview_text === 'string' && args.preview_text.trim() !== '' ? { previewText: args.preview_text.trim() } : {}),
223
- ...(typeof args.speed === 'number' ? { speed: args.speed } : {}),
224
- ...(typeof args.duration === 'number' ? { duration: args.duration } : {}),
225
- ...(typeof args.lyrics === 'string' && args.lyrics.trim() !== '' ? { lyrics: args.lyrics.trim() } : {}),
226
- ...(typeof args.is_instrumental === 'boolean' ? { isInstrumental: args.is_instrumental } : {}),
227
- ...(typeof args.loop === 'boolean' ? { loop: args.loop } : {}),
228
- ...(typeof args.prompt_influence === 'number' && Number.isFinite(args.prompt_influence) ? { promptInfluence: args.prompt_influence } : {}),
229
- ...(typeof args.seed === 'number' && Number.isFinite(args.seed) ? { seed: args.seed } : {}),
230
- ...(typeof args.steps === 'number' && Number.isFinite(args.steps) ? { steps: args.steps } : {}),
231
- ...(typeof args.cfg_scale === 'number' && Number.isFinite(args.cfg_scale) ? { cfgScale: args.cfg_scale } : {}),
232
- ...(typeof args.format === 'string' && args.format.trim() !== '' ? { format: args.format.trim() } : {}),
233
- // ---- MiniMax TTS 专属字段 ----
234
- ...(typeof args.emotion === 'string' && args.emotion.trim() !== '' ? { emotion: args.emotion.trim() } : {}),
235
- ...(typeof args.vol === 'number' && Number.isFinite(args.vol) ? { vol: args.vol } : {}),
236
- ...(typeof args.pitch === 'number' && Number.isFinite(args.pitch) ? { pitch: args.pitch } : {}),
237
- ...(typeof args.text_normalization === 'boolean' ? { textNormalization: args.text_normalization } : {}),
238
- ...(typeof args.latex_read === 'boolean' ? { latexRead: args.latex_read } : {}),
239
- ...(Array.isArray(args.pronunciation_tone) && args.pronunciation_tone.length > 0
240
- ? { pronunciationTone: args.pronunciation_tone.filter((item): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()) }
241
- : {}),
242
- ...(typeof args.sample_rate === 'number' && Number.isFinite(args.sample_rate) ? { sampleRate: args.sample_rate } : {}),
243
- ...(typeof args.bitrate === 'number' && Number.isFinite(args.bitrate) ? { bitrate: args.bitrate } : {}),
244
- ...(typeof args.channel === 'number' && Number.isFinite(args.channel) ? { audioChannel: args.channel } : {}),
245
- ...(typeof args.force_cbr === 'boolean' ? { forceCbr: args.force_cbr } : {}),
246
- ...(typeof args.subtitle_enable === 'boolean' ? { subtitleEnable: args.subtitle_enable } : {}),
247
- ...(typeof args.aigc_watermark === 'boolean' ? { aigcWatermark: args.aigc_watermark } : {}),
248
- ...(typeof args.language_boost === 'string' && args.language_boost.trim() !== '' ? { languageBoost: args.language_boost.trim() } : {}),
249
- ...(voiceModify !== undefined ? { voiceModify } : {}),
250
- ...(timbreWeights !== undefined && timbreWeights.length > 0 ? { timbreWeights } : {}),
219
+ const mode: AudioMode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
220
+ /** 把生成参数(snake_case 入参或 model_params 片段)映射为请求字段。 */
221
+ const mapParams = (raw: Record<string, unknown>): Partial<GenerateAudioRequest> => {
222
+ const voiceModify = typeof raw.voice_modify === 'object' && raw.voice_modify !== null
223
+ ? (() => {
224
+ const src = raw.voice_modify as Record<string, unknown>
225
+ const out: { pitch?: number; intensity?: number; timbre?: number; soundEffects?: string } = {}
226
+ if (typeof src.pitch === 'number') out.pitch = src.pitch
227
+ if (typeof src.intensity === 'number') out.intensity = src.intensity
228
+ if (typeof src.timbre === 'number') out.timbre = src.timbre
229
+ if (typeof src.sound_effects === 'string' && src.sound_effects.trim() !== '') out.soundEffects = src.sound_effects.trim()
230
+ return Object.keys(out).length > 0 ? out : undefined
231
+ })()
232
+ : undefined
233
+ const timbreWeights = Array.isArray(raw.timbre_weights)
234
+ ? raw.timbre_weights
235
+ .filter((item): item is { voice_id: string; weight: number } => typeof item === 'object' && item !== null && typeof (item as { voice_id?: unknown }).voice_id === 'string' && typeof (item as { weight?: unknown }).weight === 'number')
236
+ .map(item => ({ voiceId: (item.voice_id as string).trim(), weight: item.weight as number }))
237
+ .filter(item => item.voiceId !== '')
238
+ : undefined
239
+ const stringOrEmpty = (key: string): string | undefined => {
240
+ const value = raw[key]
241
+ return typeof value === 'string' && value.trim() !== '' ? value.trim() : undefined
242
+ }
243
+ const finiteOrUndefined = (key: string): number | undefined => {
244
+ const value = raw[key]
245
+ return typeof value === 'number' && Number.isFinite(value) ? value : undefined
246
+ }
247
+ return {
248
+ ...(stringOrEmpty('voice') !== undefined ? { voice: stringOrEmpty('voice')! } : {}),
249
+ ...(stringOrEmpty('preview_text') !== undefined ? { previewText: stringOrEmpty('preview_text')! } : {}),
250
+ ...(finiteOrUndefined('speed') !== undefined ? { speed: finiteOrUndefined('speed')! } : {}),
251
+ ...(finiteOrUndefined('duration') !== undefined ? { duration: finiteOrUndefined('duration')! } : {}),
252
+ ...(stringOrEmpty('lyrics') !== undefined ? { lyrics: stringOrEmpty('lyrics')! } : {}),
253
+ ...(typeof raw.is_instrumental === 'boolean' ? { isInstrumental: raw.is_instrumental } : {}),
254
+ ...(typeof raw.loop === 'boolean' ? { loop: raw.loop } : {}),
255
+ ...(finiteOrUndefined('prompt_influence') !== undefined ? { promptInfluence: finiteOrUndefined('prompt_influence')! } : {}),
256
+ ...(finiteOrUndefined('seed') !== undefined ? { seed: finiteOrUndefined('seed')! } : {}),
257
+ ...(finiteOrUndefined('steps') !== undefined ? { steps: finiteOrUndefined('steps')! } : {}),
258
+ ...(finiteOrUndefined('cfg_scale') !== undefined ? { cfgScale: finiteOrUndefined('cfg_scale')! } : {}),
259
+ ...(stringOrEmpty('format') !== undefined ? { format: stringOrEmpty('format')! } : {}),
260
+ // ---- MiniMax / ElevenLabs / Stability 专属字段 ----
261
+ ...(stringOrEmpty('emotion') !== undefined ? { emotion: stringOrEmpty('emotion')! } : {}),
262
+ ...(finiteOrUndefined('vol') !== undefined ? { vol: finiteOrUndefined('vol')! } : {}),
263
+ ...(finiteOrUndefined('pitch') !== undefined ? { pitch: finiteOrUndefined('pitch')! } : {}),
264
+ ...(typeof raw.text_normalization === 'boolean' ? { textNormalization: raw.text_normalization } : {}),
265
+ ...(typeof raw.latex_read === 'boolean' ? { latexRead: raw.latex_read } : {}),
266
+ ...(Array.isArray(raw.pronunciation_tone) && raw.pronunciation_tone.length > 0
267
+ ? { pronunciationTone: raw.pronunciation_tone.filter((item): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()) }
268
+ : {}),
269
+ ...(finiteOrUndefined('sample_rate') !== undefined ? { sampleRate: finiteOrUndefined('sample_rate')! } : {}),
270
+ ...(finiteOrUndefined('bitrate') !== undefined ? { bitrate: finiteOrUndefined('bitrate')! } : {}),
271
+ ...(finiteOrUndefined('channel') !== undefined ? { audioChannel: finiteOrUndefined('channel')! } : {}),
272
+ ...(typeof raw.force_cbr === 'boolean' ? { forceCbr: raw.force_cbr } : {}),
273
+ ...(typeof raw.subtitle_enable === 'boolean' ? { subtitleEnable: raw.subtitle_enable } : {}),
274
+ ...(typeof raw.aigc_watermark === 'boolean' ? { aigcWatermark: raw.aigc_watermark } : {}),
275
+ ...(stringOrEmpty('language_boost') !== undefined ? { languageBoost: stringOrEmpty('language_boost')! } : {}),
276
+ ...(voiceModify !== undefined ? { voiceModify } : {}),
277
+ ...(timbreWeights !== undefined && timbreWeights.length > 0 ? { timbreWeights } : {}),
278
+ }
251
279
  }
252
- try {
253
- const outputs = await generateAudio(picked.channel, request, exec.signal)
254
- const audio: AgentAudioRef[] = []
255
- const saved: SavedAudioRef[] = []
256
- for (const [index, output] of outputs.entries()) {
257
- const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`)
258
- saved.push({
259
- id: stored.id,
260
- url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
261
- file: stored.file,
262
- mime: stored.mime,
263
- bytes: stored.bytes,
264
- ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
265
- })
266
- audio.push({
267
- id: stored.id,
268
- url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
269
- mime: stored.mime,
270
- bytes: stored.bytes,
271
- ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
272
- })
280
+ const buildRequest = (picked: { channel: AudioChannel; alias: string; upstream: string }): GenerateAudioRequest => {
281
+ const base = mapParams(args as unknown as Record<string, unknown>)
282
+ // 每模型参数覆盖(model_params[alias]);缺省 = 自动沿用全局配置
283
+ let override: Partial<GenerateAudioRequest> = {}
284
+ if (typeof args.model_params === 'object' && args.model_params !== null) {
285
+ const perModel = (args.model_params as Record<string, unknown>)[picked.alias]
286
+ if (typeof perModel === 'object' && perModel !== null) override = mapParams(perModel as Record<string, unknown>)
273
287
  }
288
+ return {
289
+ mode,
290
+ model: picked.alias,
291
+ upstream: picked.upstream,
292
+ channelId: picked.channel.id,
293
+ channel: picked.channel.name,
294
+ prompt: typeof args.prompt === 'string' ? args.prompt.trim() : '',
295
+ ...base,
296
+ ...override,
297
+ }
298
+ }
299
+ /** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
300
+ const runOne = async (picked: { channel: AudioChannel; alias: string; upstream: string }): Promise<AgentAudioGroup> => {
301
+ const request = buildRequest(picked)
274
302
  try {
275
- await appendHistory({
276
- id: randomUUID(),
277
- createdAt: Date.now(),
278
- mode: request.mode,
279
- model: picked.alias,
280
- prompt: request.prompt,
281
- ...(request.voice === undefined ? {} : { voice: request.voice }),
282
- ...(request.speed === undefined ? {} : { speed: request.speed }),
283
- ...(request.duration === undefined ? {} : { duration: request.duration }),
284
- ...(request.format === undefined ? {} : { format: request.format }),
285
- audio: outputs.map((output, index) => ({
286
- id: saved[index]!.id,
287
- file: saved[index]!.file,
288
- b64: Buffer.from(output.data).toString('base64'),
289
- mime: saved[index]!.mime,
290
- bytes: saved[index]!.bytes,
291
- url: saved[index]!.url,
303
+ const outputs = await generateAudio(picked.channel, request, exec.signal)
304
+ const audio: AgentAudioRef[] = []
305
+ const saved: SavedAudioRef[] = []
306
+ for (const [index, output] of outputs.entries()) {
307
+ const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`)
308
+ saved.push({
309
+ id: stored.id,
310
+ url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
311
+ file: stored.file,
312
+ mime: stored.mime,
313
+ bytes: stored.bytes,
292
314
  ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
293
- })),
294
- channelId: picked.channel.id,
295
- channel: picked.channel.name,
296
- params: { ...request },
297
- })
298
- } catch {
299
- // History is best-effort and must not fail the agent tool.
300
- }
301
- // ---- 资源库保存:显式参数优先;设置自动入库时可用 false 跳过 ----
302
- const wantSave = args.save_to_library === true || (config.autoSaveToLibrary && args.save_to_library !== false)
303
- let resources: string[] | undefined
304
- if (wantSave) {
315
+ })
316
+ audio.push({
317
+ id: stored.id,
318
+ url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
319
+ mime: stored.mime,
320
+ bytes: stored.bytes,
321
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
322
+ })
323
+ }
305
324
  try {
306
- const entry = await saveToLibrary({
307
- audioFiles: saved.map(item => ({
308
- id: item.id,
309
- file: item.file,
310
- mime: item.mime,
311
- ...(item.voiceId === undefined ? {} : { voiceId: item.voiceId }),
325
+ await appendHistory({
326
+ id: randomUUID(),
327
+ createdAt: Date.now(),
328
+ mode: request.mode,
329
+ model: picked.alias,
330
+ prompt: request.prompt,
331
+ ...(request.voice === undefined ? {} : { voice: request.voice }),
332
+ ...(request.speed === undefined ? {} : { speed: request.speed }),
333
+ ...(request.duration === undefined ? {} : { duration: request.duration }),
334
+ ...(request.format === undefined ? {} : { format: request.format }),
335
+ audio: outputs.map((output, index) => ({
336
+ id: saved[index]!.id,
337
+ file: saved[index]!.file,
338
+ b64: Buffer.from(output.data).toString('base64'),
339
+ mime: saved[index]!.mime,
340
+ bytes: saved[index]!.bytes,
341
+ url: saved[index]!.url,
342
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
312
343
  })),
313
- type: libraryTypeOf(request.mode, args.library_type),
314
- ...(typeof args.library_name === 'string' && args.library_name.trim() !== '' ? { name: args.library_name.trim() } : {}),
315
- ...(Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag): tag is string => typeof tag === 'string' && tag.trim() !== '').map(tag => tag.trim()) } : {}),
316
- provenance: {
317
- mode: request.mode,
318
- prompt: request.prompt,
319
- channel: picked.channel.name,
320
- channelId: picked.channel.id,
321
- apiUrl: picked.channel.apiUrl,
322
- model: picked.alias,
323
- upstream: picked.upstream,
324
- ...(request.voice === undefined ? {} : { voice: request.voice }),
325
- params: { ...request },
326
- },
344
+ channelId: picked.channel.id,
345
+ channel: picked.channel.name,
346
+ params: { ...request },
327
347
  })
328
- resources = [entry.id]
329
348
  } catch {
330
- // library-save is best-effort; generation already succeeded.
349
+ // History is best-effort and must not fail the agent tool.
350
+ }
351
+ // ---- 资源库保存:显式参数优先;设置自动入库时可用 false 跳过 ----
352
+ const wantSave = args.save_to_library === true || (config.autoSaveToLibrary && args.save_to_library !== false)
353
+ let resources: string[] | undefined
354
+ if (wantSave) {
355
+ try {
356
+ const entry = await saveToLibrary({
357
+ audioFiles: saved.map(item => ({
358
+ id: item.id,
359
+ file: item.file,
360
+ mime: item.mime,
361
+ ...(item.voiceId === undefined ? {} : { voiceId: item.voiceId }),
362
+ })),
363
+ type: libraryTypeOf(request.mode, args.library_type),
364
+ ...(typeof args.library_name === 'string' && args.library_name.trim() !== '' ? { name: args.library_name.trim() } : {}),
365
+ ...(Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag): tag is string => typeof tag === 'string' && tag.trim() !== '').map(tag => tag.trim()) } : {}),
366
+ provenance: {
367
+ mode: request.mode,
368
+ prompt: request.prompt,
369
+ channel: picked.channel.name,
370
+ channelId: picked.channel.id,
371
+ apiUrl: picked.channel.apiUrl,
372
+ model: picked.alias,
373
+ upstream: picked.upstream,
374
+ ...(request.voice === undefined ? {} : { voice: request.voice }),
375
+ params: { ...request },
376
+ },
377
+ })
378
+ resources = [entry.id]
379
+ } catch {
380
+ // library-save is best-effort; generation already succeeded.
381
+ }
382
+ }
383
+ return {
384
+ model: picked.alias,
385
+ audio,
386
+ ...(resources === undefined ? {} : { resources }),
387
+ }
388
+ } catch (error) {
389
+ if (exec.signal?.aborted === true) throw error
390
+ return {
391
+ model: picked.alias,
392
+ audio: [],
393
+ error: error instanceof Error ? error.message : String(error),
331
394
  }
332
395
  }
396
+ }
397
+ // ---- 多模型对比:同一 prompt 依次生成 ----
398
+ const requestedModels = Array.isArray(args.models)
399
+ ? [...new Set(args.models.filter((item: unknown): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()))]
400
+ : []
401
+ if (requestedModels.length > 0 && mode !== 'voice_design') {
402
+ const groups: AgentAudioGroup[] = []
403
+ let succeeded = 0
404
+ for (const alias of requestedModels) {
405
+ let picked
406
+ try {
407
+ picked = resolveModel(config, alias)
408
+ } catch (error) {
409
+ groups.push({ model: alias, audio: [], error: error instanceof Error ? error.message : String(error) })
410
+ continue
411
+ }
412
+ const group = await runOne(picked)
413
+ groups.push(group)
414
+ if (group.error === undefined) succeeded++
415
+ }
333
416
  return {
334
- status: 'completed',
335
- message: 'Audio generation completed. The audio files can be played/downloaded from the returned URLs.',
336
- mode: request.mode,
337
- model: picked.alias,
338
- audio,
339
- ...(resources === undefined ? {} : { resources }),
417
+ status: succeeded > 0 ? 'completed' : 'failed',
418
+ message: succeeded > 0
419
+ ? `Generated ${succeeded}/${groups.length} model(s) with the same prompt for comparison. The audio files can be played/downloaded from the returned URLs.`
420
+ : 'All model generations failed.',
421
+ mode,
422
+ model: groups[0]?.model ?? requestedModels[0]!,
423
+ audio: groups.flatMap(group => group.audio),
424
+ groups,
425
+ ...(succeeded === 0
426
+ ? { error: groups.map(group => `${group.model}: ${group.error ?? ''}`).filter(item => !item.endsWith(': ')).join(';') }
427
+ : {}),
340
428
  }
341
- } catch (error) {
342
- if (exec.signal?.aborted === true) throw error
429
+ }
430
+ // ---- 单模型(含音色设计) ----
431
+ const picked = mode === 'voice_design'
432
+ ? (() => {
433
+ const usable = config.channels.filter(channel => channel.apiUrl.trim() !== '' && channel.apiKey.trim() !== '')
434
+ const target = usable.find(channel => channel.id === config.defaultChannelId) ?? usable[0]
435
+ if (target === undefined) throw new AudioGenError('No usable audio channel is configured for voice design.', 'no-channel-available')
436
+ return { channel: target, alias: '', upstream: '' }
437
+ })()
438
+ : resolveModel(config, args.model)
439
+ const one = await runOne(picked)
440
+ if (one.error !== undefined) {
343
441
  return {
344
442
  status: 'failed',
345
443
  message: 'Audio generation failed.',
346
- mode: request.mode,
347
- model: picked.alias,
444
+ mode,
445
+ model: one.model,
348
446
  audio: [],
349
- error: error instanceof Error ? error.message : String(error),
447
+ error: one.error,
350
448
  }
351
449
  }
450
+ return {
451
+ status: 'completed',
452
+ message: 'Audio generation completed. The audio files can be played/downloaded from the returned URLs.',
453
+ mode,
454
+ model: one.model,
455
+ audio: one.audio,
456
+ ...(one.resources === undefined ? {} : { resources: one.resources }),
457
+ }
352
458
  },
353
459
  }))
354
460
 
@@ -805,3 +805,90 @@
805
805
  @media (prefers-reduced-motion: reduce) {
806
806
  .toast { animation-duration: 1ms; }
807
807
  }
808
+
809
+ /* 多模型对比 */
810
+ .modelCheckList {
811
+ max-height: 180px;
812
+ overflow-y: auto;
813
+ display: flex;
814
+ flex-direction: column;
815
+ gap: 4px;
816
+ padding: 8px;
817
+ border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
818
+ border-radius: 8px;
819
+ background: var(--dsw-alias-bg-layer-3, #fff);
820
+ }
821
+
822
+ .resultGroups {
823
+ display: flex;
824
+ flex-direction: column;
825
+ gap: 16px;
826
+ }
827
+
828
+ .resultGroup {
829
+ display: flex;
830
+ flex-direction: column;
831
+ gap: 8px;
832
+ }
833
+
834
+ .resultGroupHead {
835
+ display: flex;
836
+ align-items: center;
837
+ gap: 8px;
838
+ flex-wrap: wrap;
839
+ }
840
+
841
+ .resultGroupChip {
842
+ font-size: 12px;
843
+ font-weight: 600;
844
+ padding: 3px 10px;
845
+ border-radius: 999px;
846
+ background: var(--dsw-alias-bg-hover, #f3f4f6);
847
+ border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
848
+ color: var(--dsw-alias-label-primary, #1f2328);
849
+ }
850
+
851
+ .resultGroupError {
852
+ font-size: 12px;
853
+ color: #dc2626;
854
+ }
855
+
856
+ .resultGroupCount {
857
+ font-size: 11px;
858
+ color: var(--dsw-alias-label-tertiary, #9ca3af);
859
+ }
860
+
861
+ /* 每模型参数覆盖矩阵 */
862
+ .overrideTable {
863
+ display: flex;
864
+ flex-direction: column;
865
+ gap: 6px;
866
+ }
867
+
868
+ .overrideRow {
869
+ display: flex;
870
+ gap: 6px;
871
+ align-items: center;
872
+ }
873
+
874
+ .overrideCell {
875
+ flex: 1;
876
+ min-width: 0;
877
+ font-size: 12px;
878
+ font-weight: 500;
879
+ color: var(--dsw-alias-label-secondary, #6b7280);
880
+ overflow: hidden;
881
+ text-overflow: ellipsis;
882
+ white-space: nowrap;
883
+ }
884
+
885
+ .overrideCell .input {
886
+ min-height: 30px;
887
+ padding: 5px 8px;
888
+ }
889
+
890
+ .overrideCellHead {
891
+ font-weight: 700;
892
+ color: var(--dsw-alias-label-primary, #1f2328);
893
+ white-space: normal;
894
+ }