dsh-audiogen 0.3.5 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-audiogen",
3
3
  "description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
4
- "version": "0.3.5",
4
+ "version": "0.4.1",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-voice-design
3
+ description: DSH AI 音频插件(dsh-audiogen)的音色设计技能:调用 generate_audio(mode=voice_design) 并指定厂商/渠道——MiniMax(POST /v1/voice_design,prompt + preview_text)或 ElevenLabs(POST /v1/text-to-voice/design,voice_description + 试听文本 100-1000 字符,过短自动生成;返回 previews[].audio_base_64 与 generated_voice_id 供后续 TTS 复用)。
4
+ whenToUse: 用户请求设计/创建新音色、音色试听,或触发 /audio:design 时使用。
5
+ ---
1
6
  # 音色/音效设计
2
7
 
3
8
  ## 触发
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-music
3
+ description: DSH AI 音频插件(dsh-audiogen)的音乐生成技能:调用 generate_audio(mode=music);覆盖 MiniMax(music-3.0/2.6/cover,lyrics 歌词、is_instrumental 纯音乐、audio_setting 采样率 16000-44100/码率 32000-256000/格式 mp3-wav-pcm、时长)、ElevenLabs(/v1/music,music_v2,时长 3-600s、lyrics_text、force_instrumental)与 Stability(stable-audio 2/2.5/3,官方 v2beta 或 OpenAI 兼容 /v1/audio/speech 双通道,seed/steps/cfg_scale/duration)。
4
+ whenToUse: 用户请求生成音乐、配乐、BGM、纯音乐、歌曲,或触发 /audio:music 时使用;MiniMax 未给歌词且未要求纯音乐时,先补一段歌词或设置 is_instrumental。
5
+ ---
1
6
  # 音乐生成
2
7
 
3
8
  ## 触发
@@ -50,3 +55,27 @@
50
55
  | seed / generation_mode / finetune_* | — | 高级字段,暂未透出 |
51
56
 
52
57
  > 响应为音频字节流(audio/*,常为 mp3)。请求同时携带 `xi-api-key` 与 `Authorization: Bearer`,以兼容 New API 类网关(官方站任一头即可)。
58
+
59
+ ## Stability Stable Audio(官方 v2beta,multipart/form-data)
60
+
61
+ 官方端点(文本到音频,TTS 描述 / 音乐 / 音效统一走该接口,不同模型参数不同):
62
+ - `POST /v2beta/audio/stable-audio/text-to-audio` → 模型 `stable-audio-3`(202 异步,随后轮询 GET /v2beta/audio/results/{id})
63
+ - `POST /v2beta/audio/stable-audio-2/text-to-audio` → 模型 `stable-audio-2` / `stable-audio-2.5`(同步返回音频)
64
+
65
+ | 字段 | 工具/面板参数 | 说明 |
66
+ | --- | --- | --- |
67
+ | prompt | prompt | 必填,描述性提示词(乐器/情绪/风格/体裁,≤10000 字符) |
68
+ | model | model | stable-audio-3 / stable-audio-2.5 / stable-audio-2 |
69
+ | duration | duration | 秒数:3 ≤380(默认 190);2/2.5 ≤190(默认 190) |
70
+ | seed | seed | 0-4294967294,默认 0=随机;同参数同 seed 可复现 |
71
+ | steps | steps | 采样步数:2 → 30-100(默认 50);2.5/3 → 4-8(默认 8) |
72
+ | cfg_scale | cfg_scale | 1-25:2 默认 7,2.5/3 默认 1;越高越贴提示词 |
73
+ | output_format | format | mp3 / wav |
74
+
75
+ > 引擎按模型自动收敛步数/时长区间;渠道 preset/apiUrl 含 `stability` 或模型名以 `stable-audio-` 开头即走官方协议(自定义渠道同样适用)。
76
+
77
+ ### 双通道(自动选择)
78
+ - **官方 v2beta**:apiUrl 为 `https://api.stability.ai`(含 `/v2beta`、`/v2beta/audio` 形态)→ multipart 原生端点(2/2.5 同步、3 异步轮询)。
79
+ - **OpenAI 兼容网关**:apiUrl 以 `/v1` 结尾或含 `/audio/speech`(如 New API)→ `POST {apiUrl}/audio/speech`,JSON:
80
+ `{ "model": "stable-audio-2.5", "input": "<prompt>", "output_format": "mp3", "duration": 30, "seed": 0, "steps": 8, "cfg_scale": 1 }`(网关把该模型映射到 Stable 上游)。
81
+ - 一方返回 `404 Invalid URL`(未路由)时自动换另一方重试;参数在两种通道均按模型收敛(duration/seed/steps/cfg_scale/output_format)。
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-sfx
3
+ description: DSH AI 音频插件(dsh-audiogen)的音效生成技能:调用 generate_audio(mode=sfx);覆盖 ElevenLabs(/v1/sound-generation,eleven_text_to_sound_v2,loop 无缝循环、prompt_influence 0-1、duration_seconds 0.5-30)与 MiniMax、Stability 等渠道的对应字段与常见错误处理。
4
+ whenToUse: 用户请求生成音效、提示音、环境音、UI 音,或触发 /audio:sfx 时使用。
5
+ ---
1
6
  # 音效生成
2
7
 
3
8
  ## 触发
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-tts
3
+ description: DSH AI 音频插件(dsh-audiogen)的 TTS 文本转语音技能:先确认渠道/模型/音色,再调用 generate_audio(mode=tts);包含 MiniMax 官方 t2a_v2 全字段(语速/音量/音调/情绪/采样率/码率/声道/发音词典/字幕/变声/双音色混合)、ElevenLabs 与 Stable Audio 的对应参数说明,以及常见错误(voice-required、网关 404 Invalid URL 等)的处理。
4
+ whenToUse: 用户提出朗读、配音、语音合成、TTS,或触发 /audio:tts 时使用;MiniMax 必须提供音色 voice_id。
5
+ ---
1
6
  # TTS 文本转语音
2
7
 
3
8
  ## 触发
@@ -9,14 +9,15 @@ import { defineTool } from '@deepseek-ai/dsh-tools'
9
9
  import { randomUUID } from 'node:crypto'
10
10
  import type { AudioChannel } from './audio-engine.ts'
11
11
  import { generateAudio, AudioGenError } from './audio-engine.ts'
12
- import { appendHistory, saveAudioFile } from './audio-store.ts'
13
- import type { AudioMode, GenerateAudioRequest } from './protocol.ts'
12
+ import { appendHistory, saveAudioFile, saveToLibrary, listLibrary } from './audio-store.ts'
13
+ import type { AudioMode, GenerateAudioRequest, LibraryType } from './protocol.ts'
14
14
 
15
15
  export interface AgentAudioToolConfig {
16
16
  enabled: boolean
17
17
  allowAgentAudioGeneration: boolean
18
18
  channels: AudioChannel[]
19
19
  defaultChannelId: string
20
+ autoSaveToLibrary: boolean
20
21
  }
21
22
 
22
23
  interface AgentAudioRef {
@@ -27,12 +28,29 @@ interface AgentAudioRef {
27
28
  voiceId?: string
28
29
  }
29
30
 
31
+ /** Internal: the persisted file name, needed for library copies. */
32
+ interface SavedAudioRef extends AgentAudioRef {
33
+ file: string
34
+ }
35
+
36
+ /** Per-model group in a multi-model comparison result. */
37
+ interface AgentAudioGroup {
38
+ model: string
39
+ audio: AgentAudioRef[]
40
+ resources?: string[]
41
+ error?: string
42
+ }
43
+
30
44
  interface AgentAudioResult {
31
45
  status: string
32
46
  message: string
33
47
  mode: AudioMode
34
48
  model: string
35
49
  audio: AgentAudioRef[]
50
+ /** Resource-library entry ids when the audio was saved to the library. */
51
+ resources?: string[]
52
+ /** Per-model results when several models were generated with the same prompt. */
53
+ groups?: AgentAudioGroup[]
36
54
  error?: string
37
55
  }
38
56
 
@@ -48,6 +66,17 @@ const audioRefSchema = {
48
66
  },
49
67
  } as const
50
68
 
69
+ const groupSchema = {
70
+ type: 'object',
71
+ additionalProperties: false,
72
+ properties: {
73
+ model: { type: 'string', required: true },
74
+ audio: { type: 'array', required: true, items: audioRefSchema },
75
+ resources: { type: 'array', items: { type: 'string' } },
76
+ error: { type: 'string' },
77
+ },
78
+ } as const
79
+
51
80
  const resultSchema = {
52
81
  type: 'object',
53
82
  additionalProperties: false,
@@ -57,6 +86,8 @@ const resultSchema = {
57
86
  mode: { type: 'string', required: true, enum: ['tts', 'music', 'sfx', 'voice_design'] },
58
87
  model: { type: 'string', required: true },
59
88
  audio: { type: 'array', required: true, items: audioRefSchema },
89
+ resources: { type: 'array', items: { type: 'string' } },
90
+ groups: { type: 'array', items: groupSchema },
60
91
  error: { type: 'string' },
61
92
  },
62
93
  } as const
@@ -96,15 +127,31 @@ function ensureConfigured(config: AgentAudioToolConfig): void {
96
127
  if (!usable) throw new AudioGenError('Audio API credentials are not configured. Open Settings > Plugins > AI Audio, add a channel and fill its API URL and API key.', 'audio-api-not-configured')
97
128
  }
98
129
 
130
+ /** Library type from the generation mode, with an explicit override. */
131
+ function libraryTypeOf(mode: AudioMode, override: unknown): LibraryType {
132
+ if (override === 'voice' || override === 'music' || override === 'sfx' || override === 'tts') return override
133
+ if (mode === 'voice_design') return 'voice'
134
+ return mode
135
+ }
136
+
99
137
  /** Register the Agent audio tool. */
100
138
  export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioToolConfig): () => void {
101
139
  const disposer = ctx.tools.register(defineTool({
102
- name: 'generate_audio',
103
- description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and voice design (MiniMax /v1/voice_design, ElevenLabs /v1/text-to-voice/design). The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
140
+ name: 'generate_audio', description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and voice design (MiniMax /v1/voice_design, ElevenLabs /v1/text-to-voice/design). The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
104
141
  parameters: {
105
142
  prompt: { type: 'string', required: true, description: 'For tts, the text to speak. For music/sfx, a descriptive prompt.' },
106
143
  mode: { type: 'string', enum: ['tts', 'music', 'sfx', 'voice_design'], description: 'Generation mode. Defaults to tts.' },
107
144
  model: { type: 'string', description: 'One of the configured audio models/voices. Defaults to the first configured model.' },
145
+ models: {
146
+ type: 'array',
147
+ items: { type: 'string' },
148
+ description: 'Optional: several configured model aliases to generate the SAME prompt with each one, sequentially, for comparison (e.g. ["speech-2.8-hd","speech-2.6-hd"]). Cannot be combined with model; when present, models wins.',
149
+ },
150
+ model_params: {
151
+ type: 'object',
152
+ additionalProperties: true,
153
+ description: 'Optional per-model parameter overrides used with "models" (automatic by default = all models share the global params). Keys are model aliases; values are partial param objects using the same param names (format, duration, voice, speed, emotion, vol, pitch, sample_rate, bitrate, lyrics, is_instrumental, loop, prompt_influence, seed, steps, cfg_scale, subtitle_enable, aigc_watermark, language_boost, pronunciation_tone, voice_modify, timbre_weights). Unset fields fall back to the global values.',
154
+ },
108
155
  voice: { type: 'string', description: 'Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio.' },
109
156
  preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
110
157
  speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
@@ -113,6 +160,9 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
113
160
  is_instrumental: { type: 'boolean', description: 'Generate purely instrumental music without vocals/lyrics (MiniMax is_instrumental). When true, lyrics may be omitted.' },
114
161
  loop: { type: 'boolean', description: 'Create a seamlessly looping sound effect (ElevenLabs sound generation loop, only for eleven_text_to_sound_v2).' },
115
162
  prompt_influence: { type: 'number', description: 'Sound effect prompt influence 0-1 (ElevenLabs prompt_influence, default 0.3): higher follows the prompt more closely, lower is more variable.' },
163
+ seed: { type: 'integer', description: 'Stable Audio random seed 0-4294967294 (default 0 = random); same seed yields reproducible audio.' },
164
+ steps: { type: 'integer', description: 'Stable Audio sampling steps, model-dependent: stable-audio-2 30-100, stable-audio-2.5/3 4-8 (out-of-range auto-clamped).' },
165
+ cfg_scale: { type: 'number', description: 'Stable Audio prompt adherence 1-25 (stable-audio-2 default 7, 2.5/3 default 1); higher follows the prompt more strictly.' },
116
166
  format: { type: 'string', description: 'Output format such as mp3 or wav. MiniMax music supports mp3/wav/pcm.' },
117
167
  // ---- MiniMax TTS only (ignored by other providers) ----
118
168
  emotion: { type: 'string', description: 'MiniMax TTS emotion, e.g. happy/sad/angry/nervous/fearful/bored (voice_setting.emotion).' },
@@ -145,13 +195,17 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
145
195
  type: 'object',
146
196
  additionalProperties: false,
147
197
  properties: {
148
- voice_id: { type: 'string' },
149
- weight: { type: 'integer' },
198
+ voice_id: { type: 'string', required: true },
199
+ weight: { type: 'integer', required: true },
150
200
  },
151
- required: ['voice_id', 'weight'],
152
201
  },
153
202
  description: 'MiniMax TTS dual-voice blend weights (timbre_weights).',
154
203
  },
204
+ // ---- resource library ----
205
+ save_to_library: { type: 'boolean', description: 'Save the generated audio into the local resource library after success. Also enabled globally by the "auto save to library" setting; pass false to skip a single run.' },
206
+ library_name: { type: 'string', description: 'Resource name in the library. Defaults to the prompt.' },
207
+ library_type: { type: 'string', enum: ['voice', 'music', 'sfx', 'tts'], description: 'Resource type in the library. Defaults to the generation mode (voice_design → voice).' },
208
+ library_tags: { type: 'array', items: { type: 'string' }, description: 'Tags for the library resource.' },
155
209
  },
156
210
  output: {
157
211
  schema: resultSchema,
@@ -162,7 +216,218 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
162
216
  async execute(args, exec) {
163
217
  const config = resolve()
164
218
  ensureConfigured(config)
165
- const mode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
219
+ const mode: AudioMode = args.mode === 'music' ? 'music' : args.mode === 'sfx' ? 'sfx' : args.mode === 'voice_design' ? 'voice_design' : 'tts'
220
+ /** 把生成参数(snake_case 入参或 model_params 片段)映射为请求字段。 */
221
+ const mapParams = (raw: Record<string, unknown>): Partial<GenerateAudioRequest> => {
222
+ const voiceModify = typeof raw.voice_modify === 'object' && raw.voice_modify !== null
223
+ ? (() => {
224
+ const src = raw.voice_modify as Record<string, unknown>
225
+ const out: { pitch?: number; intensity?: number; timbre?: number; soundEffects?: string } = {}
226
+ if (typeof src.pitch === 'number') out.pitch = src.pitch
227
+ if (typeof src.intensity === 'number') out.intensity = src.intensity
228
+ if (typeof src.timbre === 'number') out.timbre = src.timbre
229
+ if (typeof src.sound_effects === 'string' && src.sound_effects.trim() !== '') out.soundEffects = src.sound_effects.trim()
230
+ return Object.keys(out).length > 0 ? out : undefined
231
+ })()
232
+ : undefined
233
+ const timbreWeights = Array.isArray(raw.timbre_weights)
234
+ ? raw.timbre_weights
235
+ .filter((item): item is { voice_id: string; weight: number } => typeof item === 'object' && item !== null && typeof (item as { voice_id?: unknown }).voice_id === 'string' && typeof (item as { weight?: unknown }).weight === 'number')
236
+ .map(item => ({ voiceId: (item.voice_id as string).trim(), weight: item.weight as number }))
237
+ .filter(item => item.voiceId !== '')
238
+ : undefined
239
+ const stringOrEmpty = (key: string): string | undefined => {
240
+ const value = raw[key]
241
+ return typeof value === 'string' && value.trim() !== '' ? value.trim() : undefined
242
+ }
243
+ const finiteOrUndefined = (key: string): number | undefined => {
244
+ const value = raw[key]
245
+ return typeof value === 'number' && Number.isFinite(value) ? value : undefined
246
+ }
247
+ return {
248
+ ...(stringOrEmpty('voice') !== undefined ? { voice: stringOrEmpty('voice')! } : {}),
249
+ ...(stringOrEmpty('preview_text') !== undefined ? { previewText: stringOrEmpty('preview_text')! } : {}),
250
+ ...(finiteOrUndefined('speed') !== undefined ? { speed: finiteOrUndefined('speed')! } : {}),
251
+ ...(finiteOrUndefined('duration') !== undefined ? { duration: finiteOrUndefined('duration')! } : {}),
252
+ ...(stringOrEmpty('lyrics') !== undefined ? { lyrics: stringOrEmpty('lyrics')! } : {}),
253
+ ...(typeof raw.is_instrumental === 'boolean' ? { isInstrumental: raw.is_instrumental } : {}),
254
+ ...(typeof raw.loop === 'boolean' ? { loop: raw.loop } : {}),
255
+ ...(finiteOrUndefined('prompt_influence') !== undefined ? { promptInfluence: finiteOrUndefined('prompt_influence')! } : {}),
256
+ ...(finiteOrUndefined('seed') !== undefined ? { seed: finiteOrUndefined('seed')! } : {}),
257
+ ...(finiteOrUndefined('steps') !== undefined ? { steps: finiteOrUndefined('steps')! } : {}),
258
+ ...(finiteOrUndefined('cfg_scale') !== undefined ? { cfgScale: finiteOrUndefined('cfg_scale')! } : {}),
259
+ ...(stringOrEmpty('format') !== undefined ? { format: stringOrEmpty('format')! } : {}),
260
+ // ---- MiniMax / ElevenLabs / Stability 专属字段 ----
261
+ ...(stringOrEmpty('emotion') !== undefined ? { emotion: stringOrEmpty('emotion')! } : {}),
262
+ ...(finiteOrUndefined('vol') !== undefined ? { vol: finiteOrUndefined('vol')! } : {}),
263
+ ...(finiteOrUndefined('pitch') !== undefined ? { pitch: finiteOrUndefined('pitch')! } : {}),
264
+ ...(typeof raw.text_normalization === 'boolean' ? { textNormalization: raw.text_normalization } : {}),
265
+ ...(typeof raw.latex_read === 'boolean' ? { latexRead: raw.latex_read } : {}),
266
+ ...(Array.isArray(raw.pronunciation_tone) && raw.pronunciation_tone.length > 0
267
+ ? { pronunciationTone: raw.pronunciation_tone.filter((item): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()) }
268
+ : {}),
269
+ ...(finiteOrUndefined('sample_rate') !== undefined ? { sampleRate: finiteOrUndefined('sample_rate')! } : {}),
270
+ ...(finiteOrUndefined('bitrate') !== undefined ? { bitrate: finiteOrUndefined('bitrate')! } : {}),
271
+ ...(finiteOrUndefined('channel') !== undefined ? { audioChannel: finiteOrUndefined('channel')! } : {}),
272
+ ...(typeof raw.force_cbr === 'boolean' ? { forceCbr: raw.force_cbr } : {}),
273
+ ...(typeof raw.subtitle_enable === 'boolean' ? { subtitleEnable: raw.subtitle_enable } : {}),
274
+ ...(typeof raw.aigc_watermark === 'boolean' ? { aigcWatermark: raw.aigc_watermark } : {}),
275
+ ...(stringOrEmpty('language_boost') !== undefined ? { languageBoost: stringOrEmpty('language_boost')! } : {}),
276
+ ...(voiceModify !== undefined ? { voiceModify } : {}),
277
+ ...(timbreWeights !== undefined && timbreWeights.length > 0 ? { timbreWeights } : {}),
278
+ }
279
+ }
280
+ const buildRequest = (picked: { channel: AudioChannel; alias: string; upstream: string }): GenerateAudioRequest => {
281
+ const base = mapParams(args as unknown as Record<string, unknown>)
282
+ // 每模型参数覆盖(model_params[alias]);缺省 = 自动沿用全局配置
283
+ let override: Partial<GenerateAudioRequest> = {}
284
+ if (typeof args.model_params === 'object' && args.model_params !== null) {
285
+ const perModel = (args.model_params as Record<string, unknown>)[picked.alias]
286
+ if (typeof perModel === 'object' && perModel !== null) override = mapParams(perModel as Record<string, unknown>)
287
+ }
288
+ return {
289
+ mode,
290
+ model: picked.alias,
291
+ upstream: picked.upstream,
292
+ channelId: picked.channel.id,
293
+ channel: picked.channel.name,
294
+ prompt: typeof args.prompt === 'string' ? args.prompt.trim() : '',
295
+ ...base,
296
+ ...override,
297
+ }
298
+ }
299
+ /** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
300
+ const runOne = async (picked: { channel: AudioChannel; alias: string; upstream: string }): Promise<AgentAudioGroup> => {
301
+ const request = buildRequest(picked)
302
+ try {
303
+ const outputs = await generateAudio(picked.channel, request, exec.signal)
304
+ const audio: AgentAudioRef[] = []
305
+ const saved: SavedAudioRef[] = []
306
+ for (const [index, output] of outputs.entries()) {
307
+ const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`)
308
+ saved.push({
309
+ id: stored.id,
310
+ url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
311
+ file: stored.file,
312
+ mime: stored.mime,
313
+ bytes: stored.bytes,
314
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
315
+ })
316
+ audio.push({
317
+ id: stored.id,
318
+ url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
319
+ mime: stored.mime,
320
+ bytes: stored.bytes,
321
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
322
+ })
323
+ }
324
+ try {
325
+ await appendHistory({
326
+ id: randomUUID(),
327
+ createdAt: Date.now(),
328
+ mode: request.mode,
329
+ model: picked.alias,
330
+ prompt: request.prompt,
331
+ ...(request.voice === undefined ? {} : { voice: request.voice }),
332
+ ...(request.speed === undefined ? {} : { speed: request.speed }),
333
+ ...(request.duration === undefined ? {} : { duration: request.duration }),
334
+ ...(request.format === undefined ? {} : { format: request.format }),
335
+ audio: outputs.map((output, index) => ({
336
+ id: saved[index]!.id,
337
+ file: saved[index]!.file,
338
+ b64: Buffer.from(output.data).toString('base64'),
339
+ mime: saved[index]!.mime,
340
+ bytes: saved[index]!.bytes,
341
+ url: saved[index]!.url,
342
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
343
+ })),
344
+ channelId: picked.channel.id,
345
+ channel: picked.channel.name,
346
+ params: { ...request },
347
+ })
348
+ } catch {
349
+ // History is best-effort and must not fail the agent tool.
350
+ }
351
+ // ---- 资源库保存:显式参数优先;设置自动入库时可用 false 跳过 ----
352
+ const wantSave = args.save_to_library === true || (config.autoSaveToLibrary && args.save_to_library !== false)
353
+ let resources: string[] | undefined
354
+ if (wantSave) {
355
+ try {
356
+ const entry = await saveToLibrary({
357
+ audioFiles: saved.map(item => ({
358
+ id: item.id,
359
+ file: item.file,
360
+ mime: item.mime,
361
+ ...(item.voiceId === undefined ? {} : { voiceId: item.voiceId }),
362
+ })),
363
+ type: libraryTypeOf(request.mode, args.library_type),
364
+ ...(typeof args.library_name === 'string' && args.library_name.trim() !== '' ? { name: args.library_name.trim() } : {}),
365
+ ...(Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag): tag is string => typeof tag === 'string' && tag.trim() !== '').map(tag => tag.trim()) } : {}),
366
+ provenance: {
367
+ mode: request.mode,
368
+ prompt: request.prompt,
369
+ channel: picked.channel.name,
370
+ channelId: picked.channel.id,
371
+ apiUrl: picked.channel.apiUrl,
372
+ model: picked.alias,
373
+ upstream: picked.upstream,
374
+ ...(request.voice === undefined ? {} : { voice: request.voice }),
375
+ params: { ...request },
376
+ },
377
+ })
378
+ resources = [entry.id]
379
+ } catch {
380
+ // library-save is best-effort; generation already succeeded.
381
+ }
382
+ }
383
+ return {
384
+ model: picked.alias,
385
+ audio,
386
+ ...(resources === undefined ? {} : { resources }),
387
+ }
388
+ } catch (error) {
389
+ if (exec.signal?.aborted === true) throw error
390
+ return {
391
+ model: picked.alias,
392
+ audio: [],
393
+ error: error instanceof Error ? error.message : String(error),
394
+ }
395
+ }
396
+ }
397
+ // ---- 多模型对比:同一 prompt 依次生成 ----
398
+ const requestedModels = Array.isArray(args.models)
399
+ ? [...new Set(args.models.filter((item: unknown): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()))]
400
+ : []
401
+ if (requestedModels.length > 0 && mode !== 'voice_design') {
402
+ const groups: AgentAudioGroup[] = []
403
+ let succeeded = 0
404
+ for (const alias of requestedModels) {
405
+ let picked
406
+ try {
407
+ picked = resolveModel(config, alias)
408
+ } catch (error) {
409
+ groups.push({ model: alias, audio: [], error: error instanceof Error ? error.message : String(error) })
410
+ continue
411
+ }
412
+ const group = await runOne(picked)
413
+ groups.push(group)
414
+ if (group.error === undefined) succeeded++
415
+ }
416
+ return {
417
+ status: succeeded > 0 ? 'completed' : 'failed',
418
+ message: succeeded > 0
419
+ ? `Generated ${succeeded}/${groups.length} model(s) with the same prompt for comparison. The audio files can be played/downloaded from the returned URLs.`
420
+ : 'All model generations failed.',
421
+ mode,
422
+ model: groups[0]?.model ?? requestedModels[0]!,
423
+ audio: groups.flatMap(group => group.audio),
424
+ groups,
425
+ ...(succeeded === 0
426
+ ? { error: groups.map(group => `${group.model}: ${group.error ?? ''}`).filter(item => !item.endsWith(': ')).join(';') }
427
+ : {}),
428
+ }
429
+ }
430
+ // ---- 单模型(含音色设计) ----
166
431
  const picked = mode === 'voice_design'
167
432
  ? (() => {
168
433
  const usable = config.channels.filter(channel => channel.apiUrl.trim() !== '' && channel.apiKey.trim() !== '')
@@ -171,115 +436,97 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
171
436
  return { channel: target, alias: '', upstream: '' }
172
437
  })()
173
438
  : resolveModel(config, args.model)
174
- const voiceModify = typeof args.voice_modify === 'object' && args.voice_modify !== null
175
- ? (() => {
176
- const raw = args.voice_modify as Record<string, unknown>
177
- const out: { pitch?: number; intensity?: number; timbre?: number; soundEffects?: string } = {}
178
- if (typeof raw.pitch === 'number') out.pitch = raw.pitch
179
- if (typeof raw.intensity === 'number') out.intensity = raw.intensity
180
- if (typeof raw.timbre === 'number') out.timbre = raw.timbre
181
- if (typeof raw.sound_effects === 'string' && raw.sound_effects.trim() !== '') out.soundEffects = raw.sound_effects.trim()
182
- return Object.keys(out).length > 0 ? out : undefined
183
- })()
184
- : undefined
185
- const timbreWeights = Array.isArray(args.timbre_weights)
186
- ? args.timbre_weights
187
- .filter((item): item is { voice_id: string; weight: number } => typeof item === 'object' && item !== null && typeof (item as { voice_id?: unknown }).voice_id === 'string' && typeof (item as { weight?: unknown }).weight === 'number')
188
- .map(item => ({ voiceId: (item.voice_id as string).trim(), weight: item.weight as number }))
189
- .filter(item => item.voiceId !== '')
190
- : undefined
191
- const request: GenerateAudioRequest = {
192
- mode,
193
- model: picked.alias,
194
- upstream: picked.upstream,
195
- channelId: picked.channel.id,
196
- channel: picked.channel.name,
197
- prompt: args.prompt.trim(),
198
- ...(typeof args.voice === 'string' && args.voice.trim() !== '' ? { voice: args.voice.trim() } : {}),
199
- ...(typeof args.preview_text === 'string' && args.preview_text.trim() !== '' ? { previewText: args.preview_text.trim() } : {}),
200
- ...(typeof args.speed === 'number' ? { speed: args.speed } : {}),
201
- ...(typeof args.duration === 'number' ? { duration: args.duration } : {}),
202
- ...(typeof args.lyrics === 'string' && args.lyrics.trim() !== '' ? { lyrics: args.lyrics.trim() } : {}),
203
- ...(typeof args.is_instrumental === 'boolean' ? { isInstrumental: args.is_instrumental } : {}),
204
- ...(typeof args.loop === 'boolean' ? { loop: args.loop } : {}),
205
- ...(typeof args.prompt_influence === 'number' && Number.isFinite(args.prompt_influence) ? { promptInfluence: args.prompt_influence } : {}),
206
- ...(typeof args.format === 'string' && args.format.trim() !== '' ? { format: args.format.trim() } : {}),
207
- // ---- MiniMax TTS 专属字段 ----
208
- ...(typeof args.emotion === 'string' && args.emotion.trim() !== '' ? { emotion: args.emotion.trim() } : {}),
209
- ...(typeof args.vol === 'number' && Number.isFinite(args.vol) ? { vol: args.vol } : {}),
210
- ...(typeof args.pitch === 'number' && Number.isFinite(args.pitch) ? { pitch: args.pitch } : {}),
211
- ...(typeof args.text_normalization === 'boolean' ? { textNormalization: args.text_normalization } : {}),
212
- ...(typeof args.latex_read === 'boolean' ? { latexRead: args.latex_read } : {}),
213
- ...(Array.isArray(args.pronunciation_tone) && args.pronunciation_tone.length > 0
214
- ? { pronunciationTone: args.pronunciation_tone.filter((item): item is string => typeof item === 'string' && item.trim() !== '').map((item: string) => item.trim()) }
215
- : {}),
216
- ...(typeof args.sample_rate === 'number' && Number.isFinite(args.sample_rate) ? { sampleRate: args.sample_rate } : {}),
217
- ...(typeof args.bitrate === 'number' && Number.isFinite(args.bitrate) ? { bitrate: args.bitrate } : {}),
218
- ...(typeof args.channel === 'number' && Number.isFinite(args.channel) ? { audioChannel: args.channel } : {}),
219
- ...(typeof args.force_cbr === 'boolean' ? { forceCbr: args.force_cbr } : {}),
220
- ...(typeof args.subtitle_enable === 'boolean' ? { subtitleEnable: args.subtitle_enable } : {}),
221
- ...(typeof args.aigc_watermark === 'boolean' ? { aigcWatermark: args.aigc_watermark } : {}),
222
- ...(typeof args.language_boost === 'string' && args.language_boost.trim() !== '' ? { languageBoost: args.language_boost.trim() } : {}),
223
- ...(voiceModify !== undefined ? { voiceModify } : {}),
224
- ...(timbreWeights !== undefined && timbreWeights.length > 0 ? { timbreWeights } : {}),
225
- }
226
- try {
227
- const outputs = await generateAudio(picked.channel, request, exec.signal)
228
- const audio: AgentAudioRef[] = []
229
- for (const [index, output] of outputs.entries()) {
230
- const saved = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`)
231
- audio.push({
232
- id: saved.id,
233
- url: `/api/dsh-audiogen/audio/${encodeURIComponent(saved.file)}`,
234
- mime: saved.mime,
235
- bytes: saved.bytes,
236
- ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
237
- })
238
- }
239
- try {
240
- await appendHistory({
241
- id: randomUUID(),
242
- createdAt: Date.now(),
243
- mode: request.mode,
244
- model: picked.alias,
245
- prompt: request.prompt,
246
- ...(request.voice === undefined ? {} : { voice: request.voice }),
247
- ...(request.speed === undefined ? {} : { speed: request.speed }),
248
- ...(request.duration === undefined ? {} : { duration: request.duration }),
249
- ...(request.format === undefined ? {} : { format: request.format }),
250
- audio: outputs.map((output, index) => ({
251
- id: audio[index]!.id,
252
- b64: Buffer.from(output.data).toString('base64'),
253
- mime: audio[index]!.mime,
254
- bytes: audio[index]!.bytes,
255
- url: audio[index]!.url,
256
- ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
257
- })),
258
- channelId: picked.channel.id,
259
- channel: picked.channel.name,
260
- })
261
- } catch {
262
- // History is best-effort and must not fail the agent tool.
263
- }
264
- return {
265
- status: 'completed',
266
- message: 'Audio generation completed. The audio files can be played/downloaded from the returned URLs.',
267
- mode: request.mode,
268
- model: picked.alias,
269
- audio,
270
- }
271
- } catch (error) {
272
- if (exec.signal?.aborted === true) throw error
439
+ const one = await runOne(picked)
440
+ if (one.error !== undefined) {
273
441
  return {
274
442
  status: 'failed',
275
443
  message: 'Audio generation failed.',
276
- mode: request.mode,
277
- model: picked.alias,
444
+ mode,
445
+ model: one.model,
278
446
  audio: [],
279
- error: error instanceof Error ? error.message : String(error),
447
+ error: one.error,
280
448
  }
281
449
  }
450
+ return {
451
+ status: 'completed',
452
+ message: 'Audio generation completed. The audio files can be played/downloaded from the returned URLs.',
453
+ mode,
454
+ model: one.model,
455
+ audio: one.audio,
456
+ ...(one.resources === undefined ? {} : { resources: one.resources }),
457
+ }
458
+ },
459
+ }))
460
+
461
+ const searchDisposer = ctx.tools.register(defineTool({
462
+ name: 'search_audio_library',
463
+ description: 'Search curated audio resources in the local resource library (voice / music / sfx / tts). Returns matching resources with type, category, name, tags, full provenance (channel, model, voiceId, prompt) and same-origin audio URLs the user can play. Use it before generating to reuse an existing voice, music bed or sound effect instead of generating a new one.',
464
+ parameters: {
465
+ type: { type: 'string', enum: ['voice', 'music', 'sfx', 'tts'], description: 'Filter by resource type.' },
466
+ category: { type: 'string', description: 'Filter by category (voice: male/female/custom; tts: the speaking voice key).' },
467
+ keyword: { type: 'string', description: 'Search name, tags, prompt and model.' },
468
+ },
469
+ output: {
470
+ schema: {
471
+ type: 'object',
472
+ additionalProperties: false,
473
+ properties: {
474
+ status: { type: 'string', required: true, enum: ['ok'] },
475
+ count: { type: 'integer', required: true },
476
+ entries: {
477
+ type: 'array', required: true,
478
+ items: {
479
+ type: 'object',
480
+ additionalProperties: false,
481
+ properties: {
482
+ id: { type: 'string', required: true },
483
+ name: { type: 'string', required: true },
484
+ type: { type: 'string', required: true, enum: ['voice', 'music', 'sfx', 'tts'] },
485
+ category: { type: 'string' },
486
+ tags: { type: 'array', items: { type: 'string' }, required: true },
487
+ prompt: { type: 'string', required: true },
488
+ model: { type: 'string' },
489
+ channel: { type: 'string' },
490
+ voiceId: { type: 'string' },
491
+ urls: { type: 'array', items: { type: 'string' }, required: true },
492
+ },
493
+ },
494
+ },
495
+ },
496
+ } as const,
497
+ render: (_args, value) => [{ type: 'text', text: JSON.stringify(value) }],
498
+ },
499
+ isConcurrencySafe: () => true,
500
+ async execute(args) {
501
+ const keyword = typeof args.keyword === 'string' ? args.keyword.trim().toLowerCase() : ''
502
+ const wantedType = args.type === 'voice' || args.type === 'music' || args.type === 'sfx' || args.type === 'tts' ? args.type : undefined
503
+ const wantedCategory = typeof args.category === 'string' && args.category.trim() !== '' ? args.category.trim() : undefined
504
+ const all = await listLibrary()
505
+ const entries = all.filter(entry => {
506
+ if (wantedType !== undefined && entry.type !== wantedType) return false
507
+ if (wantedCategory !== undefined && (entry.category ?? '') !== wantedCategory) return false
508
+ if (keyword !== '') {
509
+ const haystack = [entry.name, ...entry.tags, entry.provenance.prompt, entry.provenance.model ?? '', entry.provenance.channel ?? ''].join(' ').toLowerCase()
510
+ if (!haystack.includes(keyword)) return false
511
+ }
512
+ return true
513
+ }).slice(0, 30).map(entry => ({
514
+ id: entry.id,
515
+ name: entry.name,
516
+ type: entry.type,
517
+ ...(entry.category === undefined ? {} : { category: entry.category }),
518
+ tags: entry.tags,
519
+ prompt: entry.provenance.prompt,
520
+ ...(entry.provenance.model === undefined ? {} : { model: entry.provenance.model }),
521
+ ...(entry.provenance.channel === undefined ? {} : { channel: entry.provenance.channel }),
522
+ ...(entry.provenance.voiceId === undefined ? {} : { voiceId: entry.provenance.voiceId }),
523
+ urls: entry.files.map(file => file.url),
524
+ }))
525
+ return { status: 'ok' as const, count: entries.length, entries }
282
526
  },
283
527
  }))
284
- return disposer
528
+ return () => {
529
+ disposer()
530
+ searchDisposer()
531
+ }
285
532
  }