dsh-audiogen 0.3.5 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -1
- package/lib/client.js +2357 -679
- package/lib/client.js.map +1 -1
- package/lib/index.js +883 -45
- package/package.json +1 -1
- package/skills/music/SKILL.md +24 -0
- package/src/agent-audio-tools.ts +155 -14
- package/src/audio-engine.ts +173 -11
- package/src/audio-presets.ts +4 -3
- package/src/audio-store.ts +251 -3
- package/src/client/AudioGenPanel.tsx +59 -376
- package/src/client/SettingsCard.tsx +9 -0
- package/src/client/api.ts +45 -2
- package/src/client/audio-panel.module.css +652 -79
- package/src/client/audio-player.tsx +87 -0
- package/src/client/icons.tsx +170 -0
- package/src/client/library-save-dialog.tsx +173 -0
- package/src/client/library-view.tsx +511 -0
- package/src/client/library.module.css +686 -0
- package/src/client/locales.ts +6 -0
- package/src/client/settings-scope.ts +2 -0
- package/src/client/studio-view.tsx +560 -0
- package/src/index.ts +6 -0
- package/src/protocol.ts +126 -1
- package/src/routes.ts +237 -4
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-audiogen",
|
|
3
3
|
"description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.4.0",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|
package/skills/music/SKILL.md
CHANGED
|
@@ -50,3 +50,27 @@
|
|
|
50
50
|
| seed / generation_mode / finetune_* | — | 高级字段,暂未透出 |
|
|
51
51
|
|
|
52
52
|
> 响应为音频字节流(audio/*,常为 mp3)。请求同时携带 `xi-api-key` 与 `Authorization: Bearer`,以兼容 New API 类网关(官方站任一头即可)。
|
|
53
|
+
|
|
54
|
+
## Stability Stable Audio(官方 v2beta,multipart/form-data)
|
|
55
|
+
|
|
56
|
+
官方端点(文本到音频,TTS 描述 / 音乐 / 音效统一走该接口,不同模型参数不同):
|
|
57
|
+
- `POST /v2beta/audio/stable-audio/text-to-audio` → 模型 `stable-audio-3`(202 异步,随后轮询 GET /v2beta/audio/results/{id})
|
|
58
|
+
- `POST /v2beta/audio/stable-audio-2/text-to-audio` → 模型 `stable-audio-2` / `stable-audio-2.5`(同步返回音频)
|
|
59
|
+
|
|
60
|
+
| 字段 | 工具/面板参数 | 说明 |
|
|
61
|
+
| --- | --- | --- |
|
|
62
|
+
| prompt | prompt | 必填,描述性提示词(乐器/情绪/风格/体裁,≤10000 字符) |
|
|
63
|
+
| model | model | stable-audio-3 / stable-audio-2.5 / stable-audio-2 |
|
|
64
|
+
| duration | duration | 秒数:3 ≤380(默认 190);2/2.5 ≤190(默认 190) |
|
|
65
|
+
| seed | seed | 0-4294967294,默认 0=随机;同参数同 seed 可复现 |
|
|
66
|
+
| steps | steps | 采样步数:2 → 30-100(默认 50);2.5/3 → 4-8(默认 8) |
|
|
67
|
+
| cfg_scale | cfg_scale | 1-25:2 默认 7,2.5/3 默认 1;越高越贴提示词 |
|
|
68
|
+
| output_format | format | mp3 / wav |
|
|
69
|
+
|
|
70
|
+
> 引擎按模型自动收敛步数/时长区间;渠道 preset/apiUrl 含 `stability` 或模型名以 `stable-audio-` 开头即走官方协议(自定义渠道同样适用)。
|
|
71
|
+
|
|
72
|
+
### 双通道(自动选择)
|
|
73
|
+
- **官方 v2beta**:apiUrl 为 `https://api.stability.ai`(含 `/v2beta`、`/v2beta/audio` 形态)→ multipart 原生端点(2/2.5 同步、3 异步轮询)。
|
|
74
|
+
- **OpenAI 兼容网关**:apiUrl 以 `/v1` 结尾或含 `/audio/speech`(如 New API)→ `POST {apiUrl}/audio/speech`,JSON:
|
|
75
|
+
`{ "model": "stable-audio-2.5", "input": "<prompt>", "output_format": "mp3", "duration": 30, "seed": 0, "steps": 8, "cfg_scale": 1 }`(网关把该模型映射到 Stable 上游)。
|
|
76
|
+
- 一方返回 `404 Invalid URL`(未路由)时自动换另一方重试;参数在两种通道均按模型收敛(duration/seed/steps/cfg_scale/output_format)。
|
package/src/agent-audio-tools.ts
CHANGED
|
@@ -9,14 +9,15 @@ import { defineTool } from '@deepseek-ai/dsh-tools'
|
|
|
9
9
|
import { randomUUID } from 'node:crypto'
|
|
10
10
|
import type { AudioChannel } from './audio-engine.ts'
|
|
11
11
|
import { generateAudio, AudioGenError } from './audio-engine.ts'
|
|
12
|
-
import { appendHistory, saveAudioFile } from './audio-store.ts'
|
|
13
|
-
import type { AudioMode, GenerateAudioRequest } from './protocol.ts'
|
|
12
|
+
import { appendHistory, saveAudioFile, saveToLibrary, listLibrary } from './audio-store.ts'
|
|
13
|
+
import type { AudioMode, GenerateAudioRequest, LibraryType } from './protocol.ts'
|
|
14
14
|
|
|
15
15
|
export interface AgentAudioToolConfig {
|
|
16
16
|
enabled: boolean
|
|
17
17
|
allowAgentAudioGeneration: boolean
|
|
18
18
|
channels: AudioChannel[]
|
|
19
19
|
defaultChannelId: string
|
|
20
|
+
autoSaveToLibrary: boolean
|
|
20
21
|
}
|
|
21
22
|
|
|
22
23
|
interface AgentAudioRef {
|
|
@@ -27,12 +28,19 @@ interface AgentAudioRef {
|
|
|
27
28
|
voiceId?: string
|
|
28
29
|
}
|
|
29
30
|
|
|
31
|
+
/** Internal: the persisted file name, needed for library copies. */
|
|
32
|
+
interface SavedAudioRef extends AgentAudioRef {
|
|
33
|
+
file: string
|
|
34
|
+
}
|
|
35
|
+
|
|
30
36
|
interface AgentAudioResult {
|
|
31
37
|
status: string
|
|
32
38
|
message: string
|
|
33
39
|
mode: AudioMode
|
|
34
40
|
model: string
|
|
35
41
|
audio: AgentAudioRef[]
|
|
42
|
+
/** Resource-library entry ids when the audio was saved to the library. */
|
|
43
|
+
resources?: string[]
|
|
36
44
|
error?: string
|
|
37
45
|
}
|
|
38
46
|
|
|
@@ -57,6 +65,7 @@ const resultSchema = {
|
|
|
57
65
|
mode: { type: 'string', required: true, enum: ['tts', 'music', 'sfx', 'voice_design'] },
|
|
58
66
|
model: { type: 'string', required: true },
|
|
59
67
|
audio: { type: 'array', required: true, items: audioRefSchema },
|
|
68
|
+
resources: { type: 'array', items: { type: 'string' } },
|
|
60
69
|
error: { type: 'string' },
|
|
61
70
|
},
|
|
62
71
|
} as const
|
|
@@ -96,11 +105,17 @@ function ensureConfigured(config: AgentAudioToolConfig): void {
|
|
|
96
105
|
if (!usable) throw new AudioGenError('Audio API credentials are not configured. Open Settings > Plugins > AI Audio, add a channel and fill its API URL and API key.', 'audio-api-not-configured')
|
|
97
106
|
}
|
|
98
107
|
|
|
108
|
+
/** Library type from the generation mode, with an explicit override. */
|
|
109
|
+
function libraryTypeOf(mode: AudioMode, override: unknown): LibraryType {
|
|
110
|
+
if (override === 'voice' || override === 'music' || override === 'sfx' || override === 'tts') return override
|
|
111
|
+
if (mode === 'voice_design') return 'voice'
|
|
112
|
+
return mode
|
|
113
|
+
}
|
|
114
|
+
|
|
99
115
|
/** Register the Agent audio tool. */
|
|
100
116
|
export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioToolConfig): () => void {
|
|
101
117
|
const disposer = ctx.tools.register(defineTool({
|
|
102
|
-
name: 'generate_audio',
|
|
103
|
-
description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and voice design (MiniMax /v1/voice_design, ElevenLabs /v1/text-to-voice/design). The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
|
|
118
|
+
name: 'generate_audio', description: 'Generate audio with the configured audio provider. Supports text-to-speech, music generation, sound effects and voice design (MiniMax /v1/voice_design, ElevenLabs /v1/text-to-voice/design). The tool call waits for the upstream result and returns same-origin audio URLs; pass those URLs to the user for playback or download. If multiple models are configured, first ask the user which one to use or pass model explicitly.',
|
|
104
119
|
parameters: {
|
|
105
120
|
prompt: { type: 'string', required: true, description: 'For tts, the text to speak. For music/sfx, a descriptive prompt.' },
|
|
106
121
|
mode: { type: 'string', enum: ['tts', 'music', 'sfx', 'voice_design'], description: 'Generation mode. Defaults to tts.' },
|
|
@@ -113,6 +128,9 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
113
128
|
is_instrumental: { type: 'boolean', description: 'Generate purely instrumental music without vocals/lyrics (MiniMax is_instrumental). When true, lyrics may be omitted.' },
|
|
114
129
|
loop: { type: 'boolean', description: 'Create a seamlessly looping sound effect (ElevenLabs sound generation loop, only for eleven_text_to_sound_v2).' },
|
|
115
130
|
prompt_influence: { type: 'number', description: 'Sound effect prompt influence 0-1 (ElevenLabs prompt_influence, default 0.3): higher follows the prompt more closely, lower is more variable.' },
|
|
131
|
+
seed: { type: 'integer', description: 'Stable Audio random seed 0-4294967294 (default 0 = random); same seed yields reproducible audio.' },
|
|
132
|
+
steps: { type: 'integer', description: 'Stable Audio sampling steps, model-dependent: stable-audio-2 30-100, stable-audio-2.5/3 4-8 (out-of-range auto-clamped).' },
|
|
133
|
+
cfg_scale: { type: 'number', description: 'Stable Audio prompt adherence 1-25 (stable-audio-2 default 7, 2.5/3 default 1); higher follows the prompt more strictly.' },
|
|
116
134
|
format: { type: 'string', description: 'Output format such as mp3 or wav. MiniMax music supports mp3/wav/pcm.' },
|
|
117
135
|
// ---- MiniMax TTS only (ignored by other providers) ----
|
|
118
136
|
emotion: { type: 'string', description: 'MiniMax TTS emotion, e.g. happy/sad/angry/nervous/fearful/bored (voice_setting.emotion).' },
|
|
@@ -152,6 +170,11 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
152
170
|
},
|
|
153
171
|
description: 'MiniMax TTS dual-voice blend weights (timbre_weights).',
|
|
154
172
|
},
|
|
173
|
+
// ---- resource library ----
|
|
174
|
+
save_to_library: { type: 'boolean', description: 'Save the generated audio into the local resource library after success. Also enabled globally by the "auto save to library" setting; pass false to skip a single run.' },
|
|
175
|
+
library_name: { type: 'string', description: 'Resource name in the library. Defaults to the prompt.' },
|
|
176
|
+
library_type: { type: 'string', enum: ['voice', 'music', 'sfx', 'tts'], description: 'Resource type in the library. Defaults to the generation mode (voice_design → voice).' },
|
|
177
|
+
library_tags: { type: 'array', items: { type: 'string' }, description: 'Tags for the library resource.' },
|
|
155
178
|
},
|
|
156
179
|
output: {
|
|
157
180
|
schema: resultSchema,
|
|
@@ -203,6 +226,9 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
203
226
|
...(typeof args.is_instrumental === 'boolean' ? { isInstrumental: args.is_instrumental } : {}),
|
|
204
227
|
...(typeof args.loop === 'boolean' ? { loop: args.loop } : {}),
|
|
205
228
|
...(typeof args.prompt_influence === 'number' && Number.isFinite(args.prompt_influence) ? { promptInfluence: args.prompt_influence } : {}),
|
|
229
|
+
...(typeof args.seed === 'number' && Number.isFinite(args.seed) ? { seed: args.seed } : {}),
|
|
230
|
+
...(typeof args.steps === 'number' && Number.isFinite(args.steps) ? { steps: args.steps } : {}),
|
|
231
|
+
...(typeof args.cfg_scale === 'number' && Number.isFinite(args.cfg_scale) ? { cfgScale: args.cfg_scale } : {}),
|
|
206
232
|
...(typeof args.format === 'string' && args.format.trim() !== '' ? { format: args.format.trim() } : {}),
|
|
207
233
|
// ---- MiniMax TTS 专属字段 ----
|
|
208
234
|
...(typeof args.emotion === 'string' && args.emotion.trim() !== '' ? { emotion: args.emotion.trim() } : {}),
|
|
@@ -226,13 +252,22 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
226
252
|
try {
|
|
227
253
|
const outputs = await generateAudio(picked.channel, request, exec.signal)
|
|
228
254
|
const audio: AgentAudioRef[] = []
|
|
255
|
+
const saved: SavedAudioRef[] = []
|
|
229
256
|
for (const [index, output] of outputs.entries()) {
|
|
230
|
-
const
|
|
257
|
+
const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`)
|
|
258
|
+
saved.push({
|
|
259
|
+
id: stored.id,
|
|
260
|
+
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
261
|
+
file: stored.file,
|
|
262
|
+
mime: stored.mime,
|
|
263
|
+
bytes: stored.bytes,
|
|
264
|
+
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
265
|
+
})
|
|
231
266
|
audio.push({
|
|
232
|
-
id:
|
|
233
|
-
url: `/api/dsh-audiogen/audio/${encodeURIComponent(
|
|
234
|
-
mime:
|
|
235
|
-
bytes:
|
|
267
|
+
id: stored.id,
|
|
268
|
+
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
269
|
+
mime: stored.mime,
|
|
270
|
+
bytes: stored.bytes,
|
|
236
271
|
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
237
272
|
})
|
|
238
273
|
}
|
|
@@ -248,25 +283,60 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
248
283
|
...(request.duration === undefined ? {} : { duration: request.duration }),
|
|
249
284
|
...(request.format === undefined ? {} : { format: request.format }),
|
|
250
285
|
audio: outputs.map((output, index) => ({
|
|
251
|
-
id:
|
|
286
|
+
id: saved[index]!.id,
|
|
287
|
+
file: saved[index]!.file,
|
|
252
288
|
b64: Buffer.from(output.data).toString('base64'),
|
|
253
|
-
mime:
|
|
254
|
-
bytes:
|
|
255
|
-
url:
|
|
289
|
+
mime: saved[index]!.mime,
|
|
290
|
+
bytes: saved[index]!.bytes,
|
|
291
|
+
url: saved[index]!.url,
|
|
256
292
|
...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
|
|
257
293
|
})),
|
|
258
294
|
channelId: picked.channel.id,
|
|
259
295
|
channel: picked.channel.name,
|
|
296
|
+
params: { ...request },
|
|
260
297
|
})
|
|
261
298
|
} catch {
|
|
262
299
|
// History is best-effort and must not fail the agent tool.
|
|
263
300
|
}
|
|
301
|
+
// ---- 资源库保存:显式参数优先;设置自动入库时可用 false 跳过 ----
|
|
302
|
+
const wantSave = args.save_to_library === true || (config.autoSaveToLibrary && args.save_to_library !== false)
|
|
303
|
+
let resources: string[] | undefined
|
|
304
|
+
if (wantSave) {
|
|
305
|
+
try {
|
|
306
|
+
const entry = await saveToLibrary({
|
|
307
|
+
audioFiles: saved.map(item => ({
|
|
308
|
+
id: item.id,
|
|
309
|
+
file: item.file,
|
|
310
|
+
mime: item.mime,
|
|
311
|
+
...(item.voiceId === undefined ? {} : { voiceId: item.voiceId }),
|
|
312
|
+
})),
|
|
313
|
+
type: libraryTypeOf(request.mode, args.library_type),
|
|
314
|
+
...(typeof args.library_name === 'string' && args.library_name.trim() !== '' ? { name: args.library_name.trim() } : {}),
|
|
315
|
+
...(Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag): tag is string => typeof tag === 'string' && tag.trim() !== '').map(tag => tag.trim()) } : {}),
|
|
316
|
+
provenance: {
|
|
317
|
+
mode: request.mode,
|
|
318
|
+
prompt: request.prompt,
|
|
319
|
+
channel: picked.channel.name,
|
|
320
|
+
channelId: picked.channel.id,
|
|
321
|
+
apiUrl: picked.channel.apiUrl,
|
|
322
|
+
model: picked.alias,
|
|
323
|
+
upstream: picked.upstream,
|
|
324
|
+
...(request.voice === undefined ? {} : { voice: request.voice }),
|
|
325
|
+
params: { ...request },
|
|
326
|
+
},
|
|
327
|
+
})
|
|
328
|
+
resources = [entry.id]
|
|
329
|
+
} catch {
|
|
330
|
+
// library-save is best-effort; generation already succeeded.
|
|
331
|
+
}
|
|
332
|
+
}
|
|
264
333
|
return {
|
|
265
334
|
status: 'completed',
|
|
266
335
|
message: 'Audio generation completed. The audio files can be played/downloaded from the returned URLs.',
|
|
267
336
|
mode: request.mode,
|
|
268
337
|
model: picked.alias,
|
|
269
338
|
audio,
|
|
339
|
+
...(resources === undefined ? {} : { resources }),
|
|
270
340
|
}
|
|
271
341
|
} catch (error) {
|
|
272
342
|
if (exec.signal?.aborted === true) throw error
|
|
@@ -281,5 +351,76 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
281
351
|
}
|
|
282
352
|
},
|
|
283
353
|
}))
|
|
284
|
-
|
|
354
|
+
|
|
355
|
+
const searchDisposer = ctx.tools.register(defineTool({
|
|
356
|
+
name: 'search_audio_library',
|
|
357
|
+
description: 'Search curated audio resources in the local resource library (voice / music / sfx / tts). Returns matching resources with type, category, name, tags, full provenance (channel, model, voiceId, prompt) and same-origin audio URLs the user can play. Use it before generating to reuse an existing voice, music bed or sound effect instead of generating a new one.',
|
|
358
|
+
parameters: {
|
|
359
|
+
type: { type: 'string', enum: ['voice', 'music', 'sfx', 'tts'], description: 'Filter by resource type.' },
|
|
360
|
+
category: { type: 'string', description: 'Filter by category (voice: male/female/custom; tts: the speaking voice key).' },
|
|
361
|
+
keyword: { type: 'string', description: 'Search name, tags, prompt and model.' },
|
|
362
|
+
},
|
|
363
|
+
output: {
|
|
364
|
+
schema: {
|
|
365
|
+
type: 'object',
|
|
366
|
+
additionalProperties: false,
|
|
367
|
+
properties: {
|
|
368
|
+
status: { type: 'string', required: true, enum: ['ok'] },
|
|
369
|
+
count: { type: 'integer', required: true },
|
|
370
|
+
entries: {
|
|
371
|
+
type: 'array', required: true,
|
|
372
|
+
items: {
|
|
373
|
+
type: 'object',
|
|
374
|
+
additionalProperties: false,
|
|
375
|
+
properties: {
|
|
376
|
+
id: { type: 'string', required: true },
|
|
377
|
+
name: { type: 'string', required: true },
|
|
378
|
+
type: { type: 'string', required: true, enum: ['voice', 'music', 'sfx', 'tts'] },
|
|
379
|
+
category: { type: 'string' },
|
|
380
|
+
tags: { type: 'array', items: { type: 'string' }, required: true },
|
|
381
|
+
prompt: { type: 'string', required: true },
|
|
382
|
+
model: { type: 'string' },
|
|
383
|
+
channel: { type: 'string' },
|
|
384
|
+
voiceId: { type: 'string' },
|
|
385
|
+
urls: { type: 'array', items: { type: 'string' }, required: true },
|
|
386
|
+
},
|
|
387
|
+
},
|
|
388
|
+
},
|
|
389
|
+
},
|
|
390
|
+
} as const,
|
|
391
|
+
render: (_args, value) => [{ type: 'text', text: JSON.stringify(value) }],
|
|
392
|
+
},
|
|
393
|
+
isConcurrencySafe: () => true,
|
|
394
|
+
async execute(args) {
|
|
395
|
+
const keyword = typeof args.keyword === 'string' ? args.keyword.trim().toLowerCase() : ''
|
|
396
|
+
const wantedType = args.type === 'voice' || args.type === 'music' || args.type === 'sfx' || args.type === 'tts' ? args.type : undefined
|
|
397
|
+
const wantedCategory = typeof args.category === 'string' && args.category.trim() !== '' ? args.category.trim() : undefined
|
|
398
|
+
const all = await listLibrary()
|
|
399
|
+
const entries = all.filter(entry => {
|
|
400
|
+
if (wantedType !== undefined && entry.type !== wantedType) return false
|
|
401
|
+
if (wantedCategory !== undefined && (entry.category ?? '') !== wantedCategory) return false
|
|
402
|
+
if (keyword !== '') {
|
|
403
|
+
const haystack = [entry.name, ...entry.tags, entry.provenance.prompt, entry.provenance.model ?? '', entry.provenance.channel ?? ''].join(' ').toLowerCase()
|
|
404
|
+
if (!haystack.includes(keyword)) return false
|
|
405
|
+
}
|
|
406
|
+
return true
|
|
407
|
+
}).slice(0, 30).map(entry => ({
|
|
408
|
+
id: entry.id,
|
|
409
|
+
name: entry.name,
|
|
410
|
+
type: entry.type,
|
|
411
|
+
...(entry.category === undefined ? {} : { category: entry.category }),
|
|
412
|
+
tags: entry.tags,
|
|
413
|
+
prompt: entry.provenance.prompt,
|
|
414
|
+
...(entry.provenance.model === undefined ? {} : { model: entry.provenance.model }),
|
|
415
|
+
...(entry.provenance.channel === undefined ? {} : { channel: entry.provenance.channel }),
|
|
416
|
+
...(entry.provenance.voiceId === undefined ? {} : { voiceId: entry.provenance.voiceId }),
|
|
417
|
+
urls: entry.files.map(file => file.url),
|
|
418
|
+
}))
|
|
419
|
+
return { status: 'ok' as const, count: entries.length, entries }
|
|
420
|
+
},
|
|
421
|
+
}))
|
|
422
|
+
return () => {
|
|
423
|
+
disposer()
|
|
424
|
+
searchDisposer()
|
|
425
|
+
}
|
|
285
426
|
}
|
package/src/audio-engine.ts
CHANGED
|
@@ -63,7 +63,13 @@ export function detectAudioMime(data: Uint8Array): string | undefined {
|
|
|
63
63
|
|
|
64
64
|
function mimeFromContentType(value: string | null): string | undefined {
|
|
65
65
|
if (value === null || value === '') return undefined
|
|
66
|
-
|
|
66
|
+
const parts = value.split(';')
|
|
67
|
+
for (const part of parts.slice(1)) {
|
|
68
|
+
// `application/json; type=audio/mpeg` —— Stability 官方用该形式携带音频真实类型。
|
|
69
|
+
const match = /^\s*type=([^;\s]+)/i.exec(part)
|
|
70
|
+
if (match !== null) return match[1]!.trim().toLowerCase()
|
|
71
|
+
}
|
|
72
|
+
return parts[0]!.trim().toLowerCase()
|
|
67
73
|
}
|
|
68
74
|
|
|
69
75
|
function audioMime(data: Uint8Array, contentType: string | null): string {
|
|
@@ -630,30 +636,183 @@ async function minimax(channel: AudioChannel, request: GenerateAudioRequest, sig
|
|
|
630
636
|
}
|
|
631
637
|
}
|
|
632
638
|
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
639
|
+
/** Stability 内部信号:路由缺失(网关 404 Invalid URL),可切换另一协议重试。 */
|
|
640
|
+
class StabilityRouteMissError extends Error {}
|
|
641
|
+
|
|
642
|
+
/** 网关风格:apiUrl 形如 .../v1、.../v1/audio/speech 时优先 OpenAI 兼容 speech。 */
|
|
643
|
+
function stabilityGatewayStyle(channel: AudioChannel): boolean {
|
|
644
|
+
const url = channel.apiUrl.trim().toLowerCase()
|
|
645
|
+
return /\/v1(\/|$|\?)/.test(url) || /\/audio\/speech(\?|$)/.test(url)
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
function isStabilityRouteMiss(status: number, detail: string): boolean {
|
|
649
|
+
return status === 404 && /invalid url|invalid_request_error/i.test(detail)
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
/**
|
|
653
|
+
* Stable Audio 官方 v2beta(multipart/form-data)。
|
|
654
|
+
* - stable-audio-3 → POST {base}/stable-audio/text-to-audio (202 异步 → GET /v2beta/audio/results/{id} 轮询)
|
|
655
|
+
* - stable-audio-2 / 2.5 → POST {base}/stable-audio-2/text-to-audio (200 同步返回音频/JSON base64)
|
|
656
|
+
* - 不同模型参数不同:stable-audio-3 steps 4-8、duration ≤380;2 steps 30-100、cfg_scale 默认 7;
|
|
657
|
+
* 2.5 steps 4-8、cfg_scale 默认 1;均支持 seed、output_format(hp3|wav)。
|
|
658
|
+
*/
|
|
659
|
+
async function stabilityNativeAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
660
|
+
const rawBase = endpointBase(channel.apiUrl)
|
|
661
|
+
const model = (request.upstream ?? request.model) || 'stable-audio-2.5'
|
|
662
|
+
const isV3 = /^stable-audio-3/i.test(model)
|
|
663
|
+
const isV2 = /^stable-audio-2(\.[05])?$/i.test(model) || /^stable-audio-2-/i.test(model)
|
|
664
|
+
const group = isV2 ? 'stable-audio-2' : 'stable-audio'
|
|
665
|
+
|
|
666
|
+
// 规范化 base:允许 apiUrl 为 `https://api.stability.ai` / `.../v2beta` / `.../v2beta/audio`
|
|
667
|
+
const base = /\/v2beta\/audio$/i.test(rawBase)
|
|
668
|
+
? rawBase
|
|
669
|
+
: /\/v2beta$/i.test(rawBase)
|
|
670
|
+
? `${rawBase}/audio`
|
|
671
|
+
: `${rawBase}/v2beta/audio`
|
|
672
|
+
const endpoint = `${base}/${group}/text-to-audio`
|
|
673
|
+
|
|
674
|
+
const form = new FormData()
|
|
675
|
+
form.set('prompt', request.prompt)
|
|
676
|
+
form.set('model', model)
|
|
677
|
+
if (request.duration !== undefined && Number.isFinite(request.duration)) {
|
|
678
|
+
const maxDuration = isV3 ? 380 : 190
|
|
679
|
+
form.set('duration', String(Math.min(maxDuration, Math.max(1, request.duration))))
|
|
680
|
+
}
|
|
681
|
+
if (request.seed !== undefined && Number.isFinite(request.seed)) {
|
|
682
|
+
form.set('seed', String(Math.floor(Math.min(4294967294, Math.max(0, request.seed)))))
|
|
683
|
+
}
|
|
684
|
+
const format = request.format === 'wav' ? 'wav' : 'mp3'
|
|
685
|
+
form.set('output_format', format)
|
|
686
|
+
if (request.steps !== undefined && Number.isInteger(request.steps)) {
|
|
687
|
+
const minSteps = isV2 && !/2\.5/i.test(model) ? 30 : 4
|
|
688
|
+
const maxSteps = isV2 && !/2\.5/i.test(model) ? 100 : 8
|
|
689
|
+
form.set('steps', String(Math.min(maxSteps, Math.max(minSteps, request.steps))))
|
|
690
|
+
}
|
|
691
|
+
if (request.cfgScale !== undefined && Number.isFinite(request.cfgScale)) {
|
|
692
|
+
form.set('cfg_scale', String(Math.min(25, Math.max(1, request.cfgScale))))
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
const response = await fetchWithTimeout(endpoint, {
|
|
696
|
+
method: 'POST',
|
|
697
|
+
redirect: 'error',
|
|
698
|
+
headers: {
|
|
699
|
+
authorization: `Bearer ${channel.apiKey.trim()}`,
|
|
700
|
+
accept: 'application/json',
|
|
701
|
+
},
|
|
702
|
+
body: form,
|
|
703
|
+
signal,
|
|
704
|
+
}, isV3 ? 60_000 : UPSTREAM_TIMEOUT_MS)
|
|
705
|
+
|
|
706
|
+
if (!response.ok) {
|
|
707
|
+
const detail = await response.text().catch(() => '')
|
|
708
|
+
if (isStabilityRouteMiss(response.status, detail)) throw new StabilityRouteMissError()
|
|
709
|
+
throw new AudioGenError(`Stable Audio API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
|
|
710
|
+
}
|
|
711
|
+
|
|
712
|
+
// stable-audio-3 异步:202 → 轮询结果
|
|
713
|
+
if (response.status === 202) {
|
|
714
|
+
const payload = await response.json().catch(() => ({})) as { id?: string }
|
|
715
|
+
if (payload.id === undefined || payload.id === '') {
|
|
716
|
+
throw new AudioGenError('Stable Audio accepted the job but returned no result id', 'audio-empty-result')
|
|
717
|
+
}
|
|
718
|
+
const apiOrigin = base.replace(/\/v2beta\/audio$/i, '')
|
|
719
|
+
const resultUrl = `${apiOrigin}/v2beta/audio/results/${encodeURIComponent(payload.id)}`
|
|
720
|
+
const deadline = Date.now() + UPSTREAM_TIMEOUT_MS
|
|
721
|
+
while (Date.now() < deadline) {
|
|
722
|
+
if (signal?.aborted === true) throw new AudioGenError('Stable Audio generation was aborted', 'audio-aborted')
|
|
723
|
+
const polled = await fetchWithTimeout(resultUrl, {
|
|
724
|
+
method: 'GET',
|
|
725
|
+
redirect: 'error',
|
|
726
|
+
headers: {
|
|
727
|
+
authorization: `Bearer ${channel.apiKey.trim()}`,
|
|
728
|
+
accept: 'application/json',
|
|
729
|
+
},
|
|
730
|
+
signal,
|
|
731
|
+
}, 60_000)
|
|
732
|
+
if (polled.ok) {
|
|
733
|
+
return normalizeAudioResponse(polled, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
734
|
+
}
|
|
735
|
+
if (polled.status === 404 || polled.status === 202) {
|
|
736
|
+
await new Promise(resolve => setTimeout(resolve, 5000))
|
|
737
|
+
continue
|
|
738
|
+
}
|
|
739
|
+
const detail = await polled.text().catch(() => '')
|
|
740
|
+
throw new AudioGenError(`Stable Audio result API error (HTTP ${polled.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
|
|
741
|
+
}
|
|
742
|
+
throw new AudioGenError('Stable Audio generation timed out waiting for the result', 'audio-timeout')
|
|
743
|
+
}
|
|
744
|
+
|
|
745
|
+
return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
746
|
+
}
|
|
747
|
+
|
|
748
|
+
/**
|
|
749
|
+
* Stable Audio 经 OpenAI 兼容网关(如 New API 的 /v1/audio/speech):
|
|
750
|
+
* 模型名映射到 Stable 上游,JSON 体为 {model, input, output_format, duration,
|
|
751
|
+
* seed, steps, cfg_scale} —— 与官方 v2beta 字段一一对应,网关负责转发。
|
|
752
|
+
*/
|
|
753
|
+
async function stabilityGatewayAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
754
|
+
const rawBase = endpointBase(channel.apiUrl)
|
|
755
|
+
const model = (request.upstream ?? request.model) || 'stable-audio-2.5'
|
|
756
|
+
const isV3 = /^stable-audio-3/i.test(model)
|
|
757
|
+
const isV2 = /^stable-audio-2(\.[05])?$/i.test(model) || /^stable-audio-2-/i.test(model)
|
|
758
|
+
const endpoint = /\/audio\/speech(\?|$)/i.test(rawBase) ? rawBase : `${rawBase}/audio/speech`
|
|
759
|
+
const format = request.format === 'wav' ? 'wav' : 'mp3'
|
|
637
760
|
const body: Record<string, unknown> = {
|
|
638
761
|
model,
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
...(request.
|
|
762
|
+
input: request.prompt,
|
|
763
|
+
output_format: format,
|
|
764
|
+
...(request.duration !== undefined && Number.isFinite(request.duration)
|
|
765
|
+
? { duration: Math.min(isV3 ? 380 : 190, Math.max(1, request.duration)) }
|
|
766
|
+
: {}),
|
|
767
|
+
...(request.seed !== undefined && Number.isFinite(request.seed)
|
|
768
|
+
? { seed: Math.floor(Math.min(4294967294, Math.max(0, request.seed))) }
|
|
769
|
+
: {}),
|
|
770
|
+
...(request.steps !== undefined && Number.isInteger(request.steps)
|
|
771
|
+
? { steps: Math.min(isV2 && !/2\.5/i.test(model) ? 100 : 8, Math.max(isV2 && !/2\.5/i.test(model) ? 30 : 4, request.steps)) }
|
|
772
|
+
: {}),
|
|
773
|
+
...(request.cfgScale !== undefined && Number.isFinite(request.cfgScale)
|
|
774
|
+
? { cfg_scale: Math.min(25, Math.max(1, request.cfgScale)) }
|
|
775
|
+
: {}),
|
|
642
776
|
}
|
|
643
777
|
const response = await fetchWithTimeout(endpoint, {
|
|
644
778
|
method: 'POST',
|
|
645
|
-
redirect: '
|
|
779
|
+
redirect: 'follow',
|
|
646
780
|
headers: {
|
|
647
781
|
authorization: `Bearer ${channel.apiKey.trim()}`,
|
|
782
|
+
accept: 'audio/*',
|
|
648
783
|
'content-type': 'application/json',
|
|
649
|
-
accept: 'application/json, audio/mpeg, audio/wav',
|
|
650
784
|
},
|
|
651
785
|
body: JSON.stringify(body),
|
|
652
786
|
signal,
|
|
653
787
|
}, UPSTREAM_TIMEOUT_MS)
|
|
788
|
+
if (!response.ok) {
|
|
789
|
+
const detail = await response.text().catch(() => '')
|
|
790
|
+
if (isStabilityRouteMiss(response.status, detail)) throw new StabilityRouteMissError()
|
|
791
|
+
throw new AudioGenError(`Stable Audio gateway API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
|
|
792
|
+
}
|
|
654
793
|
return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
|
|
655
794
|
}
|
|
656
795
|
|
|
796
|
+
/**
|
|
797
|
+
* 稳定性入口:优先官方 v2beta(api.stability.ai / v2beta 形态),
|
|
798
|
+
* 网关形态(apiUrl 以 /v1 结尾或已含 /audio/speech)优先 OpenAI 兼容;
|
|
799
|
+
* 一方返回 404 Invalid URL(未路由)时自动换另一方重试。
|
|
800
|
+
*/
|
|
801
|
+
async function stabilityAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
802
|
+
const styles: Array<'native' | 'gateway'> = stabilityGatewayStyle(channel) ? ['gateway', 'native'] : ['native', 'gateway']
|
|
803
|
+
let lastError: Error | undefined
|
|
804
|
+
for (const style of styles) {
|
|
805
|
+
try {
|
|
806
|
+
if (style === 'gateway') return await stabilityGatewayAudio(channel, request, signal)
|
|
807
|
+
return await stabilityNativeAudio(channel, request, signal)
|
|
808
|
+
} catch (error) {
|
|
809
|
+
if (!(error instanceof StabilityRouteMissError)) throw error
|
|
810
|
+
lastError = error
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
throw lastError ?? new AudioGenError('Stable Audio 渠道未配置或不可达', 'audio-api-error')
|
|
814
|
+
}
|
|
815
|
+
|
|
657
816
|
async function genericAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
|
|
658
817
|
const base = endpointBase(channel.apiUrl)
|
|
659
818
|
if (request.mode === 'tts' && !/\/generate(\?|$)/i.test(base)) {
|
|
@@ -701,7 +860,10 @@ export async function generateAudio(
|
|
|
701
860
|
|
|
702
861
|
if (isElevenLabs(channel)) return elevenLabs(channel, request, signal)
|
|
703
862
|
if (isMiniMax(channel)) return minimax(channel, request, signal)
|
|
704
|
-
|
|
863
|
+
// 稳定性渠道或模型名明确为 stable-audio-*(含自定义渠道)→ 走官方 Stable Audio 协议
|
|
864
|
+
if (isStability(channel) || /^stable-audio-/i.test(((request.upstream ?? request.model) || '').trim())) {
|
|
865
|
+
return stabilityAudio(channel, request, signal)
|
|
866
|
+
}
|
|
705
867
|
if (isOpenAICompatible(channel, request.mode)) return openAITTS(channel, request, signal)
|
|
706
868
|
return genericAudio(channel, request, signal)
|
|
707
869
|
}
|
package/src/audio-presets.ts
CHANGED
|
@@ -74,10 +74,11 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
|
|
|
74
74
|
name: 'Stability AI(stable-audio)',
|
|
75
75
|
apiUrl: 'https://api.stability.ai/v2beta/audio',
|
|
76
76
|
site: 'https://stability.ai/stable-audio',
|
|
77
|
-
hint: 'Stability AI 音乐 /
|
|
77
|
+
hint: 'Stability AI 文本到音频(TTS 描述 / 音乐 / 音效,stable-audio 系列;stable-audio-3 为异步任务)',
|
|
78
78
|
models: [
|
|
79
|
-
{ alias: 'stable-audio-
|
|
80
|
-
{ alias: 'stable-audio-
|
|
79
|
+
{ alias: 'stable-audio-3', id: 'stable-audio-3', category: 'music' },
|
|
80
|
+
{ alias: 'stable-audio-2.5', id: 'stable-audio-2.5', category: 'music' },
|
|
81
|
+
{ alias: 'stable-audio-2', id: 'stable-audio-2', category: 'music' },
|
|
81
82
|
],
|
|
82
83
|
},
|
|
83
84
|
]
|