dsh-audiogen 0.3.2 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -116,6 +116,7 @@ function findBase64Audio(value) {
116
116
  const record = value;
117
117
  for (const key of [
118
118
  "audio",
119
+ "music",
119
120
  "b64_json",
120
121
  "base64",
121
122
  "data",
@@ -423,15 +424,36 @@ async function minimax(channel, request, signal) {
423
424
  }];
424
425
  }
425
426
  if (request.mode === "music") {
427
+ const MUSIC_FORMATS = /* @__PURE__ */ new Set([
428
+ "mp3",
429
+ "wav",
430
+ "pcm"
431
+ ]);
432
+ const MUSIC_SAMPLE_RATES = /* @__PURE__ */ new Set([
433
+ 16e3,
434
+ 24e3,
435
+ 32e3,
436
+ 44100
437
+ ]);
438
+ const MUSIC_BITRATES = /* @__PURE__ */ new Set([
439
+ 32e3,
440
+ 64e3,
441
+ 128e3,
442
+ 256e3
443
+ ]);
444
+ const lyrics = request.lyrics?.trim() ?? "";
445
+ if (lyrics === "" && request.isInstrumental !== true) throw new AudioGenError("MiniMax 音乐生成需要歌词(lyrics 参数),或在「纯音乐」模式(is_instrumental=true)下生成;也可让面板/Agent 先为提示词创作一段歌词。", "lyrics-required");
426
446
  const endpoint = `${base}/music_generation`;
427
447
  const body = {
428
448
  model,
429
449
  prompt: request.prompt,
450
+ ...lyrics === "" ? {} : { lyrics },
451
+ ...request.isInstrumental !== void 0 ? { is_instrumental: request.isInstrumental } : {},
430
452
  ...request.duration !== void 0 ? { duration: request.duration } : {},
431
453
  audio_setting: {
432
- format: request.format ?? "mp3",
433
- sample_rate: 44100,
434
- bitrate: 256e3
454
+ format: MUSIC_FORMATS.has(request.format ?? "mp3") ? request.format ?? "mp3" : "mp3",
455
+ sample_rate: MUSIC_SAMPLE_RATES.has(request.sampleRate ?? 44100) ? request.sampleRate ?? 44100 : 44100,
456
+ bitrate: MUSIC_BITRATES.has(request.bitrate ?? 256e3) ? request.bitrate ?? 256e3 : 256e3
435
457
  }
436
458
  };
437
459
  const response = await fetchWithTimeout(endpoint, {
@@ -1054,6 +1076,8 @@ function parseGenerateRequest(body) {
1054
1076
  ...typeof body.previewText === "string" && body.previewText.trim() !== "" ? { previewText: body.previewText.trim() } : {},
1055
1077
  ...num(body.speed) !== void 0 ? { speed: num(body.speed) } : {},
1056
1078
  ...num(body.duration) !== void 0 ? { duration: num(body.duration) } : {},
1079
+ ...typeof body.lyrics === "string" && body.lyrics.trim() !== "" ? { lyrics: body.lyrics.trim() } : {},
1080
+ ...typeof body.isInstrumental === "boolean" ? { isInstrumental: body.isInstrumental } : {},
1057
1081
  ...typeof body.format === "string" && body.format.trim() !== "" ? { format: body.format.trim() } : {},
1058
1082
  ...typeof body.channelId === "string" && body.channelId !== "" ? { channelId: body.channelId } : {},
1059
1083
  ...str(body.emotion) !== void 0 ? { emotion: str(body.emotion) } : {},
@@ -1592,9 +1616,17 @@ function registerAgentAudioTools(ctx, resolve) {
1592
1616
  type: "number",
1593
1617
  description: "Requested duration in seconds for music/sfx."
1594
1618
  },
1619
+ lyrics: {
1620
+ type: "string",
1621
+ description: "Lyrics for music generation (MiniMax music-3.0/music-cover). Required unless is_instrumental is true. Split verses with an empty line."
1622
+ },
1623
+ is_instrumental: {
1624
+ type: "boolean",
1625
+ description: "Generate purely instrumental music without vocals/lyrics (MiniMax is_instrumental). When true, lyrics may be omitted."
1626
+ },
1595
1627
  format: {
1596
1628
  type: "string",
1597
- description: "Output format such as mp3 or wav."
1629
+ description: "Output format such as mp3 or wav. MiniMax music supports mp3/wav/pcm."
1598
1630
  },
1599
1631
  emotion: {
1600
1632
  type: "string",
@@ -1623,11 +1655,11 @@ function registerAgentAudioTools(ctx, resolve) {
1623
1655
  },
1624
1656
  sample_rate: {
1625
1657
  type: "integer",
1626
- description: "MiniMax TTS sample rate: 16000/24000/32000/44100/48000, default 32000 (audio_setting.sample_rate)."
1658
+ description: "MiniMax sample rate: music 16000/24000/32000/44100 (default 44100); tts default 32000 (audio_setting.sample_rate)."
1627
1659
  },
1628
1660
  bitrate: {
1629
1661
  type: "integer",
1630
- description: "MiniMax TTS bitrate in bps: 64000-320000, default 128000 (audio_setting.bitrate)."
1662
+ description: "MiniMax bitrate in bps: 32000/64000/128000/256000 (music default 256000, tts default 128000; audio_setting.bitrate)."
1631
1663
  },
1632
1664
  channel: {
1633
1665
  type: "integer",
@@ -1730,6 +1762,8 @@ function registerAgentAudioTools(ctx, resolve) {
1730
1762
  ...typeof args.preview_text === "string" && args.preview_text.trim() !== "" ? { previewText: args.preview_text.trim() } : {},
1731
1763
  ...typeof args.speed === "number" ? { speed: args.speed } : {},
1732
1764
  ...typeof args.duration === "number" ? { duration: args.duration } : {},
1765
+ ...typeof args.lyrics === "string" && args.lyrics.trim() !== "" ? { lyrics: args.lyrics.trim() } : {},
1766
+ ...typeof args.is_instrumental === "boolean" ? { isInstrumental: args.is_instrumental } : {},
1733
1767
  ...typeof args.format === "string" && args.format.trim() !== "" ? { format: args.format.trim() } : {},
1734
1768
  ...typeof args.emotion === "string" && args.emotion.trim() !== "" ? { emotion: args.emotion.trim() } : {},
1735
1769
  ...typeof args.vol === "number" && Number.isFinite(args.vol) ? { vol: args.vol } : {},
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-audiogen",
3
3
  "description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
4
- "version": "0.3.2",
4
+ "version": "0.3.4",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {
@@ -2,15 +2,38 @@
2
2
 
3
3
  ## 触发
4
4
  - `/audio:music <描述>`
5
- - 用户说“生成一段音乐 / 配乐 / BGM
5
+ - 用户说“生成一段音乐 / 配乐 / BGM / 纯音乐”
6
6
 
7
7
  ## 参数
8
8
  - prompt: 必填,风格/情绪/乐器/时长描述
9
- - model: 可选,已配置的音频模型
10
- - duration: 可选,秒数
11
- - format: 可选,mp3 / wav
9
+ - model: 可选,已配置的音频模型(MiniMax:music-3.0 / music-2.6 / music-cover)
10
+ - lyrics: 可选,歌词;**MiniMax music-3.0 / music-cover 必填**(除非 is_instrumental=true);多段用空行分隔
11
+ - is_instrumental: 可选,是否纯音乐(无歌词/人声),true lyrics 可留空
12
+ - duration: 可选,秒数(MiniMax 一般 5-120s)
13
+ - format: 可选,MiniMax 音乐仅 mp3 / wav / pcm
12
14
 
13
15
  ## 流程
14
- 1. 确认已配置支持音乐生成的渠道(如 Stability Audio / 自定义)。
15
- 2. 调用 `generate_audio`,mode=music。
16
- 3. 将生成的音频 URL 返回给用户。
16
+ 1. 确认已配置支持音乐生成的渠道(如 MiniMax / Stability Audio / 自定义)。
17
+ 2. 若用户未给歌词且未要求纯音乐:对 MiniMax 渠道先创作/补全一段歌词再调用。
18
+ 3. 调用 `generate_audio`,mode=music。
19
+ 4. 将生成的音频 URL 返回给用户。
20
+
21
+ ## MiniMax 官方 music_generation 字段参考(POST /v1/music_generation)
22
+
23
+ | 字段 | 工具参数 | 说明 |
24
+ | --- | --- | --- |
25
+ | model | model | music-3.0 / music-2.6 / music-cover |
26
+ | prompt | prompt | 音乐风格/情绪/乐器描述(≤3000 字) |
27
+ | lyrics | lyrics | 歌词;多段用空行分隔;纯音乐模式可留空 |
28
+ | is_instrumental | is_instrumental | 默认 false;true = 纯音乐(无歌词/人声),此时 lyrics 可省 |
29
+ | duration | duration | 生成时长(秒) |
30
+ | audio_setting.format | format | **mp3 / wav / pcm**(仅此三种) |
31
+ | audio_setting.sample_rate | sample_rate | **16000 / 24000 / 32000 / 44100**(无 48000) |
32
+ | audio_setting.bitrate | bitrate | **32000 / 64000 / 128000 / 256000** |
33
+
34
+ > 引擎对音乐 audio_setting 按上述枚举校验,超出枚举的取值自动回退默认(format=mp3、sample_rate=44100、bitrate=256000)。
35
+ > 若不加 lyrics 也未开启纯音乐,引擎会直接提示 `lyrics-required`(MiniMax 上游返回 2013 lyrics is required)。
36
+
37
+ ## 常见错误
38
+ - `lyrics-required`:MiniMax 音乐生成需要歌词,或开启纯音乐。
39
+ - `HTTP 400` 且含 `2013`:上游参数不合法,检查 lyrics / audio_setting 枚举。
@@ -109,7 +109,9 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
109
109
  preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
110
110
  speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
111
111
  duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
112
- format: { type: 'string', description: 'Output format such as mp3 or wav.' },
112
+ lyrics: { type: 'string', description: 'Lyrics for music generation (MiniMax music-3.0/music-cover). Required unless is_instrumental is true. Split verses with an empty line.' },
113
+ is_instrumental: { type: 'boolean', description: 'Generate purely instrumental music without vocals/lyrics (MiniMax is_instrumental). When true, lyrics may be omitted.' },
114
+ format: { type: 'string', description: 'Output format such as mp3 or wav. MiniMax music supports mp3/wav/pcm.' },
113
115
  // ---- MiniMax TTS only (ignored by other providers) ----
114
116
  emotion: { type: 'string', description: 'MiniMax TTS emotion, e.g. happy/sad/angry/nervous/fearful/bored (voice_setting.emotion).' },
115
117
  vol: { type: 'number', description: 'MiniMax TTS volume 0-10, default 1 (voice_setting.vol).' },
@@ -117,8 +119,8 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
117
119
  text_normalization: { type: 'boolean', description: 'MiniMax TTS text normalization switch (voice_setting.text_normalization).' },
118
120
  latex_read: { type: 'boolean', description: 'MiniMax TTS math formula reading switch (voice_setting.latex_read).' },
119
121
  pronunciation_tone: { type: 'array', items: { type: 'string' }, description: 'MiniMax TTS pronunciation dictionary tone entries, each "word/pronunciation", e.g. ["处理/(chu3)(li3)", "危险/dangerous"] (pronunciation_dict.tone).' },
120
- sample_rate: { type: 'integer', description: 'MiniMax TTS sample rate: 16000/24000/32000/44100/48000, default 32000 (audio_setting.sample_rate).' },
121
- bitrate: { type: 'integer', description: 'MiniMax TTS bitrate in bps: 64000-320000, default 128000 (audio_setting.bitrate).' },
122
+ sample_rate: { type: 'integer', description: 'MiniMax sample rate: music 16000/24000/32000/44100 (default 44100); tts default 32000 (audio_setting.sample_rate).' },
123
+ bitrate: { type: 'integer', description: 'MiniMax bitrate in bps: 32000/64000/128000/256000 (music default 256000, tts default 128000; audio_setting.bitrate).' },
122
124
  channel: { type: 'integer', description: 'MiniMax TTS audio channels: 1 or 2, default 1 (audio_setting.channel).' },
123
125
  force_cbr: { type: 'boolean', description: 'MiniMax TTS force CBR encoding (audio_setting.force_cbr).' },
124
126
  subtitle_enable: { type: 'boolean', description: 'MiniMax TTS subtitle output switch (subtitle_enable).' },
@@ -195,6 +197,8 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
195
197
  ...(typeof args.preview_text === 'string' && args.preview_text.trim() !== '' ? { previewText: args.preview_text.trim() } : {}),
196
198
  ...(typeof args.speed === 'number' ? { speed: args.speed } : {}),
197
199
  ...(typeof args.duration === 'number' ? { duration: args.duration } : {}),
200
+ ...(typeof args.lyrics === 'string' && args.lyrics.trim() !== '' ? { lyrics: args.lyrics.trim() } : {}),
201
+ ...(typeof args.is_instrumental === 'boolean' ? { isInstrumental: args.is_instrumental } : {}),
198
202
  ...(typeof args.format === 'string' && args.format.trim() !== '' ? { format: args.format.trim() } : {}),
199
203
  // ---- MiniMax TTS 专属字段 ----
200
204
  ...(typeof args.emotion === 'string' && args.emotion.trim() !== '' ? { emotion: args.emotion.trim() } : {}),
@@ -124,7 +124,7 @@ function findBase64Audio(value: unknown): string | undefined {
124
124
  }
125
125
  if (value === null || typeof value !== 'object') return undefined
126
126
  const record = value as Record<string, unknown>
127
- for (const key of ['audio', 'b64_json', 'base64', 'data', 'output', 'result', 'value']) {
127
+ for (const key of ['audio', 'music', 'b64_json', 'base64', 'data', 'output', 'result', 'value']) {
128
128
  const candidate = record[key]
129
129
  const found = findBase64Audio(candidate)
130
130
  if (found !== undefined) return found
@@ -437,15 +437,31 @@ async function minimax(channel: AudioChannel, request: GenerateAudioRequest, sig
437
437
  }
438
438
 
439
439
  if (request.mode === 'music') {
440
+ // MiniMax 音乐生成官方字段:model/lyrics/prompt/is_instrumental/duration/
441
+ // audio_setting{format, sample_rate, bitrate}。音频输出配置为固定枚举:
442
+ // format mp3|wav|pcm;sample_rate 16000|24000|32000|44100;
443
+ // bitrate 32000|64000|128000|256000,超出枚举的值回退默认。
444
+ const MUSIC_FORMATS = new Set(['mp3', 'wav', 'pcm'])
445
+ const MUSIC_SAMPLE_RATES = new Set([16000, 24000, 32000, 44100])
446
+ const MUSIC_BITRATES = new Set([32000, 64000, 128000, 256000])
447
+ const lyrics = request.lyrics?.trim() ?? ''
448
+ if (lyrics === '' && request.isInstrumental !== true) {
449
+ throw new AudioGenError(
450
+ 'MiniMax 音乐生成需要歌词(lyrics 参数),或在「纯音乐」模式(is_instrumental=true)下生成;也可让面板/Agent 先为提示词创作一段歌词。',
451
+ 'lyrics-required',
452
+ )
453
+ }
440
454
  const endpoint = `${base}/music_generation`
441
455
  const body: Record<string, unknown> = {
442
456
  model,
443
457
  prompt: request.prompt,
458
+ ...(lyrics === '' ? {} : { lyrics }),
459
+ ...(request.isInstrumental !== undefined ? { is_instrumental: request.isInstrumental } : {}),
444
460
  ...(request.duration !== undefined ? { duration: request.duration } : {}),
445
461
  audio_setting: {
446
- format: request.format ?? 'mp3',
447
- sample_rate: 44100,
448
- bitrate: 256000,
462
+ format: MUSIC_FORMATS.has(request.format ?? 'mp3') ? (request.format ?? 'mp3') : 'mp3',
463
+ sample_rate: MUSIC_SAMPLE_RATES.has(request.sampleRate ?? 44100) ? (request.sampleRate ?? 44100) : 44100,
464
+ bitrate: MUSIC_BITRATES.has(request.bitrate ?? 256000) ? (request.bitrate ?? 256000) : 256000,
449
465
  },
450
466
  }
451
467
  const response = await fetchWithTimeout(endpoint, {
@@ -60,6 +60,8 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
60
60
  const [voice, setVoice] = useState('')
61
61
  const [speed, setSpeed] = useState('')
62
62
  const [duration, setDuration] = useState('')
63
+ const [lyrics, setLyrics] = useState('')
64
+ const [instrumental, setInstrumental] = useState(false)
63
65
  const [format, setFormat] = useState('mp3')
64
66
  // MiniMax TTS 高级参数(其他厂商忽略)
65
67
  const [emotion, setEmotion] = useState('')
@@ -109,6 +111,8 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
109
111
  ...(voice.trim() !== '' ? { voice: voice.trim() } : {}),
110
112
  ...(speed.trim() !== '' ? { speed: Number(speed) } : {}),
111
113
  ...(duration.trim() !== '' ? { duration: Number(duration) } : {}),
114
+ ...(lyrics.trim() !== '' ? { lyrics: lyrics.trim() } : {}),
115
+ ...(instrumental ? { isInstrumental: true } : {}),
112
116
  ...(format.trim() !== '' ? { format: format.trim() } : {}),
113
117
  ...(emotion.trim() !== '' ? { emotion: emotion.trim() } : {}),
114
118
  ...(vol.trim() !== '' ? { vol: Number(vol) } : {}),
@@ -252,14 +256,53 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
252
256
  </label>
253
257
  ) : null}
254
258
 
259
+ {mode === 'music' ? (
260
+ <>
261
+ <label className={css.label}>
262
+ <span>歌词(纯音乐模式可留空;多段用空行分隔)</span>
263
+ <textarea className={css.textarea} value={lyrics} onChange={event => setLyrics(event.target.value)} placeholder={'第一段歌词…\n\n第二段歌词…'} />
264
+ </label>
265
+ <label className={css.checkbox}>
266
+ <input type="checkbox" checked={instrumental} onChange={event => setInstrumental(event.target.checked)} />
267
+ <span>纯音乐(无歌词/人声)is_instrumental</span>
268
+ </label>
269
+ <div className={css.row}>
270
+ <label className={css.label}>
271
+ <span>采样率</span>
272
+ <select className={css.select} value={sampleRate} onChange={event => setSampleRate(event.target.value)}>
273
+ <option value="">默认(44100)</option>
274
+ <option value="16000">16000</option>
275
+ <option value="24000">24000</option>
276
+ <option value="32000">32000</option>
277
+ <option value="44100">44100</option>
278
+ </select>
279
+ </label>
280
+ <label className={css.label}>
281
+ <span>码率 bps</span>
282
+ <select className={css.select} value={bitrate} onChange={event => setBitrate(event.target.value)}>
283
+ <option value="">默认(256000)</option>
284
+ <option value="32000">32000</option>
285
+ <option value="64000">64000</option>
286
+ <option value="128000">128000</option>
287
+ <option value="256000">256000</option>
288
+ </select>
289
+ </label>
290
+ </div>
291
+ </>
292
+ ) : null}
293
+
255
294
  {needModel ? (
256
295
  <label className={css.label}>
257
296
  <span>{tt('format.label')}</span>
258
297
  <select className={css.select} value={format} onChange={event => setFormat(event.target.value)}>
259
298
  <option value="mp3">mp3</option>
260
299
  <option value="wav">wav</option>
261
- <option value="flac">flac</option>
262
- <option value="ogg">ogg</option>
300
+ {mode === 'tts' ? (
301
+ <>
302
+ <option value="flac">flac</option>
303
+ <option value="ogg">ogg</option>
304
+ </>
305
+ ) : null}
263
306
  <option value="pcm">pcm</option>
264
307
  </select>
265
308
  </label>
@@ -12,7 +12,7 @@ export const zh = {
12
12
  'mode.voiceDesign': '音色设计',
13
13
  'prompt.placeholder': '输入文本、音乐/音效描述,或音色设计描述…',
14
14
  'prompt.required': '请输入文本或提示词',
15
- 'model.label': '模型 / 音色',
15
+ 'model.label': '模型',
16
16
  'voice.label': '音色',
17
17
  'speed.label': '语速',
18
18
  'duration.label': '时长(秒)',
@@ -113,7 +113,7 @@ export const en: Record<AudioGenKey, string> = {
113
113
  'mode.voiceDesign': 'Voice design',
114
114
  'prompt.placeholder': 'Text to speak, or a music/SFX/voice-design description…',
115
115
  'prompt.required': 'Prompt or text is required',
116
- 'model.label': 'Model / voice',
116
+ 'model.label': 'Model',
117
117
  'voice.label': 'Voice',
118
118
  'speed.label': 'Speed',
119
119
  'duration.label': 'Duration (s)',
package/src/protocol.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
9
9
 
10
10
  /** Published package version shared by the host updater and the client UI. */
11
- export const PLUGIN_VERSION = '0.3.2'
11
+ export const PLUGIN_VERSION = '0.3.4'
12
12
 
13
13
  /** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
14
14
  export const SETTINGS_API = {
@@ -114,6 +114,10 @@ export interface GenerateAudioRequest {
114
114
  speed?: number
115
115
  /** Requested duration in seconds (music/sfx). */
116
116
  duration?: number
117
+ /** MiniMax 音乐生成歌词;music-3.0 / music-cover 在非纯音乐模式下必填。 */
118
+ lyrics?: string
119
+ /** MiniMax 是否生成纯音乐(无歌词/人声);true 时 lyrics 可为空。 */
120
+ isInstrumental?: boolean
117
121
  /** Output format, e.g. mp3, wav, pcm. */
118
122
  format?: string
119
123
  /** Channel this request targets (host falls back to default). */
package/src/routes.ts CHANGED
@@ -122,6 +122,8 @@ function parseGenerateRequest(body: Record<string, unknown>): GenerateAudioReque
122
122
  ...(typeof body.previewText === 'string' && body.previewText.trim() !== '' ? { previewText: body.previewText.trim() } : {}),
123
123
  ...(num(body.speed) !== undefined ? { speed: num(body.speed)! } : {}),
124
124
  ...(num(body.duration) !== undefined ? { duration: num(body.duration)! } : {}),
125
+ ...(typeof body.lyrics === 'string' && body.lyrics.trim() !== '' ? { lyrics: body.lyrics.trim() } : {}),
126
+ ...(typeof body.isInstrumental === 'boolean' ? { isInstrumental: body.isInstrumental } : {}),
125
127
  ...(typeof body.format === 'string' && body.format.trim() !== '' ? { format: body.format.trim() } : {}),
126
128
  ...(typeof body.channelId === 'string' && body.channelId !== '' ? { channelId: body.channelId } : {}),
127
129
  // ---- MiniMax TTS 专属字段(其他厂商忽略) ----