dsh-audiogen 0.3.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -63,7 +63,13 @@ export function detectAudioMime(data: Uint8Array): string | undefined {
63
63
 
64
64
  function mimeFromContentType(value: string | null): string | undefined {
65
65
  if (value === null || value === '') return undefined
66
- return value.split(';')[0]!.trim().toLowerCase()
66
+ const parts = value.split(';')
67
+ for (const part of parts.slice(1)) {
68
+ // `application/json; type=audio/mpeg` —— Stability 官方用该形式携带音频真实类型。
69
+ const match = /^\s*type=([^;\s]+)/i.exec(part)
70
+ if (match !== null) return match[1]!.trim().toLowerCase()
71
+ }
72
+ return parts[0]!.trim().toLowerCase()
67
73
  }
68
74
 
69
75
  function audioMime(data: Uint8Array, contentType: string | null): string {
@@ -242,6 +248,114 @@ async function openAITTS(channel: AudioChannel, request: GenerateAudioRequest, s
242
248
  async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
243
249
  const base = endpointBase(channel.apiUrl)
244
250
  const model = (request.upstream ?? request.model) || 'eleven_multilingual_v2'
251
+ // 官方使用 xi-api-key;额外携带 Authorization Bearer 以兼容 New API 类网关。
252
+ const headers = {
253
+ 'xi-api-key': channel.apiKey.trim(),
254
+ authorization: `Bearer ${channel.apiKey.trim()}`,
255
+ 'content-type': 'application/json',
256
+ accept: 'audio/mpeg, application/json',
257
+ }
258
+
259
+ // ------------- ElevenLabs Voice Design(POST /v1/text-to-voice/design) -------------
260
+ // voice_description 必填;text 100-1000 字符,过短时用 auto_generate_text;
261
+ // 返回 previews[].audio_base_64 与 previews[].generated_voice_id。
262
+ if (request.mode === 'voice_design') {
263
+ const endpoint = `${base}/text-to-voice/design`
264
+ const previewText = request.previewText?.trim() ?? ''
265
+ const body: Record<string, unknown> = {
266
+ voice_description: request.prompt,
267
+ ...(previewText.length >= 100 ? { text: previewText } : { auto_generate_text: true }),
268
+ }
269
+ const response = await fetchWithTimeout(endpoint, {
270
+ method: 'POST',
271
+ redirect: 'error',
272
+ headers,
273
+ body: JSON.stringify(body),
274
+ signal,
275
+ }, UPSTREAM_TIMEOUT_MS)
276
+ if (!response.ok) {
277
+ const detail = await response.text().catch(() => '')
278
+ throw new AudioGenError(`ElevenLabs voice design API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
279
+ }
280
+ const payload = await response.json() as {
281
+ previews?: Array<{ audio_base_64?: string; generated_voice_id?: string; media_type?: string }>
282
+ }
283
+ const previews = payload.previews ?? []
284
+ if (previews.length === 0) throw new AudioGenError('ElevenLabs voice design returned no previews', 'audio-empty-result')
285
+ const outputs: Array<{ data: Uint8Array; mime: string; voiceId?: string }> = []
286
+ for (const preview of previews) {
287
+ const encoded = preview.audio_base_64?.trim() ?? ''
288
+ if (encoded === '') continue
289
+ const data = new Uint8Array(Buffer.from(encoded, 'base64'))
290
+ outputs.push({
291
+ data,
292
+ mime: preview.media_type ?? 'audio/mpeg',
293
+ ...(preview.generated_voice_id === undefined || preview.generated_voice_id === '' ? {} : { voiceId: preview.generated_voice_id }),
294
+ })
295
+ }
296
+ if (outputs.length === 0) throw new AudioGenError('ElevenLabs voice design returned no audio', 'audio-empty-result')
297
+ return outputs
298
+ }
299
+
300
+ // ------------- ElevenLabs Music(POST /v1/music) -------------
301
+ // 模型:music_v1 / music_v2;prompt 与 composition_plan 二选一(引擎用 prompt)。
302
+ if (request.mode === 'music') {
303
+ const endpoint = `${base}/music`
304
+ const musicModel = (request.upstream ?? request.model) || 'music_v1'
305
+ const body: Record<string, unknown> = {
306
+ model_id: musicModel,
307
+ prompt: request.prompt,
308
+ ...(request.duration !== undefined && Number.isFinite(request.duration)
309
+ ? { music_length_ms: Math.round(Math.min(600_000, Math.max(3_000, request.duration * 1000))) }
310
+ : {}),
311
+ ...(request.lyrics !== undefined && request.lyrics.trim() !== '' ? { lyrics_text: request.lyrics.trim() } : {}),
312
+ ...(request.isInstrumental !== undefined ? { force_instrumental: request.isInstrumental } : {}),
313
+ }
314
+ const response = await fetchWithTimeout(endpoint, {
315
+ method: 'POST',
316
+ redirect: 'follow',
317
+ headers,
318
+ body: JSON.stringify(body),
319
+ signal,
320
+ }, UPSTREAM_TIMEOUT_MS)
321
+ if (!response.ok) {
322
+ const detail = await response.text().catch(() => '')
323
+ throw new AudioGenError(`ElevenLabs music API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
324
+ }
325
+ return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
326
+ }
327
+
328
+ // ------------- ElevenLabs Sound Effects(POST /v1/sound-generation) -------------
329
+ // 官方模型:eleven_text_to_sound_v2;text 必填;duration_seconds 0.5-30;
330
+ // loop 仅该模型可用;prompt_influence 0-1(默认 0.3)。
331
+ if (request.mode === 'sfx') {
332
+ const endpoint = `${base}/sound-generation`
333
+ const sfxModel = (request.upstream ?? request.model) || 'eleven_text_to_sound_v2'
334
+ const body: Record<string, unknown> = {
335
+ text: request.prompt,
336
+ model_id: sfxModel,
337
+ ...(request.duration !== undefined && Number.isFinite(request.duration)
338
+ ? { duration_seconds: Math.min(30, Math.max(0.5, request.duration)) }
339
+ : {}),
340
+ ...(request.loop !== undefined ? { loop: request.loop } : {}),
341
+ ...(request.promptInfluence !== undefined && Number.isFinite(request.promptInfluence)
342
+ ? { prompt_influence: Math.min(1, Math.max(0, request.promptInfluence)) }
343
+ : {}),
344
+ }
345
+ const response = await fetchWithTimeout(endpoint, {
346
+ method: 'POST',
347
+ redirect: 'follow',
348
+ headers,
349
+ body: JSON.stringify(body),
350
+ signal,
351
+ }, UPSTREAM_TIMEOUT_MS)
352
+ if (!response.ok) {
353
+ const detail = await response.text().catch(() => '')
354
+ throw new AudioGenError(`ElevenLabs sound effects API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
355
+ }
356
+ return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
357
+ }
358
+
245
359
  const voiceId = (request.voice ?? request.model ?? model).trim()
246
360
  const endpoint = `${base}/text-to-speech/${encodeURIComponent(voiceId)}`
247
361
  const body: Record<string, unknown> = {
@@ -258,11 +372,7 @@ async function elevenLabs(channel: AudioChannel, request: GenerateAudioRequest,
258
372
  const response = await fetchWithTimeout(endpoint, {
259
373
  method: 'POST',
260
374
  redirect: 'error',
261
- headers: {
262
- 'xi-api-key': channel.apiKey.trim(),
263
- 'content-type': 'application/json',
264
- accept: 'audio/mpeg, application/json',
265
- },
375
+ headers,
266
376
  body: JSON.stringify(body),
267
377
  signal,
268
378
  }, UPSTREAM_TIMEOUT_MS)
@@ -526,30 +636,183 @@ async function minimax(channel: AudioChannel, request: GenerateAudioRequest, sig
526
636
  }
527
637
  }
528
638
 
529
- async function stabilityAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
530
- const base = endpointBase(channel.apiUrl)
531
- const endpoint = /\/generation(\?|$)/i.test(base) ? base : `${base}/generation`
532
- const model = (request.upstream ?? request.model) || 'stable-audio-2.0'
639
+ /** Stability 内部信号:路由缺失(网关 404 Invalid URL),可切换另一协议重试。 */
640
+ class StabilityRouteMissError extends Error {}
641
+
642
+ /** 网关风格:apiUrl 形如 .../v1、.../v1/audio/speech 时优先 OpenAI 兼容 speech。 */
643
+ function stabilityGatewayStyle(channel: AudioChannel): boolean {
644
+ const url = channel.apiUrl.trim().toLowerCase()
645
+ return /\/v1(\/|$|\?)/.test(url) || /\/audio\/speech(\?|$)/.test(url)
646
+ }
647
+
648
+ function isStabilityRouteMiss(status: number, detail: string): boolean {
649
+ return status === 404 && /invalid url|invalid_request_error/i.test(detail)
650
+ }
651
+
652
+ /**
653
+ * Stable Audio 官方 v2beta(multipart/form-data)。
654
+ * - stable-audio-3 → POST {base}/stable-audio/text-to-audio (202 异步 → GET /v2beta/audio/results/{id} 轮询)
655
+ * - stable-audio-2 / 2.5 → POST {base}/stable-audio-2/text-to-audio (200 同步返回音频/JSON base64)
656
+ * - 不同模型参数不同:stable-audio-3 steps 4-8、duration ≤380;2 steps 30-100、cfg_scale 默认 7;
657
+ * 2.5 steps 4-8、cfg_scale 默认 1;均支持 seed、output_format(hp3|wav)。
658
+ */
659
+ async function stabilityNativeAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
660
+ const rawBase = endpointBase(channel.apiUrl)
661
+ const model = (request.upstream ?? request.model) || 'stable-audio-2.5'
662
+ const isV3 = /^stable-audio-3/i.test(model)
663
+ const isV2 = /^stable-audio-2(\.[05])?$/i.test(model) || /^stable-audio-2-/i.test(model)
664
+ const group = isV2 ? 'stable-audio-2' : 'stable-audio'
665
+
666
+ // 规范化 base:允许 apiUrl 为 `https://api.stability.ai` / `.../v2beta` / `.../v2beta/audio`
667
+ const base = /\/v2beta\/audio$/i.test(rawBase)
668
+ ? rawBase
669
+ : /\/v2beta$/i.test(rawBase)
670
+ ? `${rawBase}/audio`
671
+ : `${rawBase}/v2beta/audio`
672
+ const endpoint = `${base}/${group}/text-to-audio`
673
+
674
+ const form = new FormData()
675
+ form.set('prompt', request.prompt)
676
+ form.set('model', model)
677
+ if (request.duration !== undefined && Number.isFinite(request.duration)) {
678
+ const maxDuration = isV3 ? 380 : 190
679
+ form.set('duration', String(Math.min(maxDuration, Math.max(1, request.duration))))
680
+ }
681
+ if (request.seed !== undefined && Number.isFinite(request.seed)) {
682
+ form.set('seed', String(Math.floor(Math.min(4294967294, Math.max(0, request.seed)))))
683
+ }
684
+ const format = request.format === 'wav' ? 'wav' : 'mp3'
685
+ form.set('output_format', format)
686
+ if (request.steps !== undefined && Number.isInteger(request.steps)) {
687
+ const minSteps = isV2 && !/2\.5/i.test(model) ? 30 : 4
688
+ const maxSteps = isV2 && !/2\.5/i.test(model) ? 100 : 8
689
+ form.set('steps', String(Math.min(maxSteps, Math.max(minSteps, request.steps))))
690
+ }
691
+ if (request.cfgScale !== undefined && Number.isFinite(request.cfgScale)) {
692
+ form.set('cfg_scale', String(Math.min(25, Math.max(1, request.cfgScale))))
693
+ }
694
+
695
+ const response = await fetchWithTimeout(endpoint, {
696
+ method: 'POST',
697
+ redirect: 'error',
698
+ headers: {
699
+ authorization: `Bearer ${channel.apiKey.trim()}`,
700
+ accept: 'application/json',
701
+ },
702
+ body: form,
703
+ signal,
704
+ }, isV3 ? 60_000 : UPSTREAM_TIMEOUT_MS)
705
+
706
+ if (!response.ok) {
707
+ const detail = await response.text().catch(() => '')
708
+ if (isStabilityRouteMiss(response.status, detail)) throw new StabilityRouteMissError()
709
+ throw new AudioGenError(`Stable Audio API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
710
+ }
711
+
712
+ // stable-audio-3 异步:202 → 轮询结果
713
+ if (response.status === 202) {
714
+ const payload = await response.json().catch(() => ({})) as { id?: string }
715
+ if (payload.id === undefined || payload.id === '') {
716
+ throw new AudioGenError('Stable Audio accepted the job but returned no result id', 'audio-empty-result')
717
+ }
718
+ const apiOrigin = base.replace(/\/v2beta\/audio$/i, '')
719
+ const resultUrl = `${apiOrigin}/v2beta/audio/results/${encodeURIComponent(payload.id)}`
720
+ const deadline = Date.now() + UPSTREAM_TIMEOUT_MS
721
+ while (Date.now() < deadline) {
722
+ if (signal?.aborted === true) throw new AudioGenError('Stable Audio generation was aborted', 'audio-aborted')
723
+ const polled = await fetchWithTimeout(resultUrl, {
724
+ method: 'GET',
725
+ redirect: 'error',
726
+ headers: {
727
+ authorization: `Bearer ${channel.apiKey.trim()}`,
728
+ accept: 'application/json',
729
+ },
730
+ signal,
731
+ }, 60_000)
732
+ if (polled.ok) {
733
+ return normalizeAudioResponse(polled, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
734
+ }
735
+ if (polled.status === 404 || polled.status === 202) {
736
+ await new Promise(resolve => setTimeout(resolve, 5000))
737
+ continue
738
+ }
739
+ const detail = await polled.text().catch(() => '')
740
+ throw new AudioGenError(`Stable Audio result API error (HTTP ${polled.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
741
+ }
742
+ throw new AudioGenError('Stable Audio generation timed out waiting for the result', 'audio-timeout')
743
+ }
744
+
745
+ return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
746
+ }
747
+
748
+ /**
749
+ * Stable Audio 经 OpenAI 兼容网关(如 New API 的 /v1/audio/speech):
750
+ * 模型名映射到 Stable 上游,JSON 体为 {model, input, output_format, duration,
751
+ * seed, steps, cfg_scale} —— 与官方 v2beta 字段一一对应,网关负责转发。
752
+ */
753
+ async function stabilityGatewayAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
754
+ const rawBase = endpointBase(channel.apiUrl)
755
+ const model = (request.upstream ?? request.model) || 'stable-audio-2.5'
756
+ const isV3 = /^stable-audio-3/i.test(model)
757
+ const isV2 = /^stable-audio-2(\.[05])?$/i.test(model) || /^stable-audio-2-/i.test(model)
758
+ const endpoint = /\/audio\/speech(\?|$)/i.test(rawBase) ? rawBase : `${rawBase}/audio/speech`
759
+ const format = request.format === 'wav' ? 'wav' : 'mp3'
533
760
  const body: Record<string, unknown> = {
534
761
  model,
535
- prompt: request.prompt,
536
- ...(request.duration !== undefined ? { duration: request.duration } : {}),
537
- ...(request.format !== undefined ? { output_format: request.format } : {}),
762
+ input: request.prompt,
763
+ output_format: format,
764
+ ...(request.duration !== undefined && Number.isFinite(request.duration)
765
+ ? { duration: Math.min(isV3 ? 380 : 190, Math.max(1, request.duration)) }
766
+ : {}),
767
+ ...(request.seed !== undefined && Number.isFinite(request.seed)
768
+ ? { seed: Math.floor(Math.min(4294967294, Math.max(0, request.seed))) }
769
+ : {}),
770
+ ...(request.steps !== undefined && Number.isInteger(request.steps)
771
+ ? { steps: Math.min(isV2 && !/2\.5/i.test(model) ? 100 : 8, Math.max(isV2 && !/2\.5/i.test(model) ? 30 : 4, request.steps)) }
772
+ : {}),
773
+ ...(request.cfgScale !== undefined && Number.isFinite(request.cfgScale)
774
+ ? { cfg_scale: Math.min(25, Math.max(1, request.cfgScale)) }
775
+ : {}),
538
776
  }
539
777
  const response = await fetchWithTimeout(endpoint, {
540
778
  method: 'POST',
541
- redirect: 'error',
779
+ redirect: 'follow',
542
780
  headers: {
543
781
  authorization: `Bearer ${channel.apiKey.trim()}`,
782
+ accept: 'audio/*',
544
783
  'content-type': 'application/json',
545
- accept: 'application/json, audio/mpeg, audio/wav',
546
784
  },
547
785
  body: JSON.stringify(body),
548
786
  signal,
549
787
  }, UPSTREAM_TIMEOUT_MS)
788
+ if (!response.ok) {
789
+ const detail = await response.text().catch(() => '')
790
+ if (isStabilityRouteMiss(response.status, detail)) throw new StabilityRouteMissError()
791
+ throw new AudioGenError(`Stable Audio gateway API error (HTTP ${response.status})${detail === '' ? '' : `: ${detail.slice(0, 300)}`}`, 'audio-api-error')
792
+ }
550
793
  return normalizeAudioResponse(response, { apiKey: channel.apiKey, fallbackMime: 'audio/mpeg' })
551
794
  }
552
795
 
796
+ /**
797
+ * 稳定性入口:优先官方 v2beta(api.stability.ai / v2beta 形态),
798
+ * 网关形态(apiUrl 以 /v1 结尾或已含 /audio/speech)优先 OpenAI 兼容;
799
+ * 一方返回 404 Invalid URL(未路由)时自动换另一方重试。
800
+ */
801
+ async function stabilityAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
802
+ const styles: Array<'native' | 'gateway'> = stabilityGatewayStyle(channel) ? ['gateway', 'native'] : ['native', 'gateway']
803
+ let lastError: Error | undefined
804
+ for (const style of styles) {
805
+ try {
806
+ if (style === 'gateway') return await stabilityGatewayAudio(channel, request, signal)
807
+ return await stabilityNativeAudio(channel, request, signal)
808
+ } catch (error) {
809
+ if (!(error instanceof StabilityRouteMissError)) throw error
810
+ lastError = error
811
+ }
812
+ }
813
+ throw lastError ?? new AudioGenError('Stable Audio 渠道未配置或不可达', 'audio-api-error')
814
+ }
815
+
553
816
  async function genericAudio(channel: AudioChannel, request: GenerateAudioRequest, signal?: AbortSignal): Promise<Array<{ data: Uint8Array; mime: string; voiceId?: string }>> {
554
817
  const base = endpointBase(channel.apiUrl)
555
818
  if (request.mode === 'tts' && !/\/generate(\?|$)/i.test(base)) {
@@ -591,13 +854,16 @@ export async function generateAudio(
591
854
  if (channel.apiUrl.trim() === '') throw new AudioGenError('channel API URL is not configured', 'audio-no-endpoint')
592
855
  if (channel.apiKey.trim() === '') throw new AudioGenError('channel API key is not configured', 'audio-no-key')
593
856
  if (request.prompt.trim() === '') throw new AudioGenError('audio prompt/text is required', 'audio-empty-prompt')
594
- if (request.mode === 'voice_design' && !isMiniMax(channel)) {
595
- throw new AudioGenError('音色设计当前仅支持 MiniMax 渠道', 'voice-design-unsupported')
857
+ if (request.mode === 'voice_design' && !isMiniMax(channel) && !isElevenLabs(channel)) {
858
+ throw new AudioGenError('音色设计当前仅支持 MiniMax(/v1/voice_design)与 ElevenLabs(/v1/text-to-voice/design)渠道', 'voice-design-unsupported')
596
859
  }
597
860
 
598
861
  if (isElevenLabs(channel)) return elevenLabs(channel, request, signal)
599
862
  if (isMiniMax(channel)) return minimax(channel, request, signal)
600
- if (isStability(channel)) return stabilityAudio(channel, request, signal)
863
+ // 稳定性渠道或模型名明确为 stable-audio-*(含自定义渠道)→ 走官方 Stable Audio 协议
864
+ if (isStability(channel) || /^stable-audio-/i.test(((request.upstream ?? request.model) || '').trim())) {
865
+ return stabilityAudio(channel, request, signal)
866
+ }
601
867
  if (isOpenAICompatible(channel, request.mode)) return openAITTS(channel, request, signal)
602
868
  return genericAudio(channel, request, signal)
603
869
  }
@@ -53,7 +53,7 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
53
53
  name: 'ElevenLabs',
54
54
  apiUrl: 'https://api.elevenlabs.io/v1',
55
55
  site: 'https://elevenlabsai.cn',
56
- hint: 'ElevenLabs 语音合成(TTS);建议点击「获取可用模型」拉取音色与模型',
56
+ hint: 'ElevenLabs 语音合成(TTS)与音乐生成(POST /v1/music,music_v2);可点「获取可用模型」拉取音色与模型',
57
57
  models: [
58
58
  { alias: 'Rachel', id: '21m00Tcm4TlvDq8ikWAM', category: 'tts' },
59
59
  { alias: 'Adam', id: 'pNInz6obpgDQGcFmaJgB', category: 'tts' },
@@ -62,6 +62,11 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
62
62
  { alias: 'eleven_multilingual_v2', id: 'eleven_multilingual_v2', category: 'tts' },
63
63
  { alias: 'eleven_turbo_v2_5', id: 'eleven_turbo_v2_5', category: 'tts' },
64
64
  { alias: 'eleven_flash_v2_5', id: 'eleven_flash_v2_5', category: 'tts' },
65
+ // ElevenLabs Music(POST /v1/music)
66
+ { alias: 'music_v2', id: 'music_v2', category: 'music' },
67
+ { alias: 'music_v1', id: 'music_v1', category: 'music' },
68
+ // ElevenLabs Sound Effects(POST /v1/sound-generation)
69
+ { alias: 'eleven_text_to_sound_v2', id: 'eleven_text_to_sound_v2', category: 'sfx' },
65
70
  ],
66
71
  },
67
72
  {
@@ -69,10 +74,11 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
69
74
  name: 'Stability AI(stable-audio)',
70
75
  apiUrl: 'https://api.stability.ai/v2beta/audio',
71
76
  site: 'https://stability.ai/stable-audio',
72
- hint: 'Stability AI 音乐 / 音效生成(stable-audio 系列)',
77
+ hint: 'Stability AI 文本到音频(TTS 描述 / 音乐 / 音效,stable-audio 系列;stable-audio-3 为异步任务)',
73
78
  models: [
74
- { alias: 'stable-audio-2.0', id: 'stable-audio-2.0', category: 'music' },
75
- { alias: 'stable-audio-1.0', id: 'stable-audio-1.0', category: 'music' },
79
+ { alias: 'stable-audio-3', id: 'stable-audio-3', category: 'music' },
80
+ { alias: 'stable-audio-2.5', id: 'stable-audio-2.5', category: 'music' },
81
+ { alias: 'stable-audio-2', id: 'stable-audio-2', category: 'music' },
76
82
  ],
77
83
  },
78
84
  ]
@@ -1,14 +1,16 @@
1
1
  /**
2
2
  * Host-side persistence for generated audio and generation history.
3
3
  * Files live under ~/.dsh/dsh-audiogen/audio/; history is one JSON document.
4
+ * The resource library lives under ~/.dsh/dsh-audiogen/library/ with one
5
+ * index JSON plus files organized by type (voice/music/sfx/tts) and category.
4
6
  */
5
7
 
6
- import { mkdir, readFile, writeFile, readdir, unlink } from 'node:fs/promises'
8
+ import { mkdir, readFile, writeFile, unlink, rename, rmdir } from 'node:fs/promises'
7
9
  import { randomUUID } from 'node:crypto'
8
10
  import path from 'node:path'
9
11
  import os from 'node:os'
10
- import type { HistoryEntry, HistoryEntryInput } from './protocol.ts'
11
- import { HISTORY_MAX } from './protocol.ts'
12
+ import type { HistoryEntry, HistoryEntryInput, LibraryEntry, LibraryType, LibraryFileRef, LibraryAudioInput, LibraryProvenance } from './protocol.ts'
13
+ import { HISTORY_MAX, LIBRARY_API, LIBRARY_NAME_MAX } from './protocol.ts'
12
14
 
13
15
  function dshHome(): string {
14
16
  return process.env.DSH_HOME ?? path.join(os.homedir(), '.dsh')
@@ -16,6 +18,8 @@ function dshHome(): string {
16
18
 
17
19
  export const AUDIO_DATA_DIR = path.join(dshHome(), 'dsh-audiogen', 'audio')
18
20
  const HISTORY_FILE = path.join(dshHome(), 'dsh-audiogen', 'history.json')
21
+ export const LIBRARY_DATA_DIR = path.join(dshHome(), 'dsh-audiogen', 'library')
22
+ const LIBRARY_INDEX_FILE = path.join(LIBRARY_DATA_DIR, 'index.json')
19
23
 
20
24
  async function ensureDir(): Promise<void> {
21
25
  await mkdir(AUDIO_DATA_DIR, { recursive: true })
@@ -111,6 +115,7 @@ export async function appendHistory(entry: HistoryEntryInput): Promise<HistoryEn
111
115
  })),
112
116
  ...(entry.channelId === undefined ? {} : { channelId: entry.channelId }),
113
117
  ...(entry.channel === undefined ? {} : { channel: entry.channel }),
118
+ ...(entry.params === undefined ? {} : { params: entry.params }),
114
119
  }, ...list].slice(0, HISTORY_MAX)
115
120
  await writeHistory(next)
116
121
  return next
@@ -131,3 +136,246 @@ export async function clearHistory(): Promise<HistoryEntry[]> {
131
136
  await writeHistory([])
132
137
  return []
133
138
  }
139
+
140
+ // ---------------------------------------------------------------------------
141
+ // Resource library
142
+ // ---------------------------------------------------------------------------
143
+
144
+ /** Library type dir names (whitelisted on the audio route too). */
145
+ const LIBRARY_TYPE_DIRS: Record<LibraryType, string> = {
146
+ voice: 'voice',
147
+ music: 'music',
148
+ sfx: 'sfx',
149
+ tts: 'tts',
150
+ }
151
+
152
+ /** Sanitize one path segment (cid or voice key). Falls back to 'default'. */
153
+ export function sanitizeSegment(value: string): string {
154
+ const cleaned = value
155
+ .replace(/[^a-zA-Z0-9\u4e00-\u9fa5._-]+/g, '_')
156
+ .replace(/^[._-]+|[._-]+$/g, '')
157
+ .slice(0, 60)
158
+ return cleaned === '' ? 'default' : cleaned
159
+ }
160
+
161
+ /** Infer the category for a save when the client did not provide one. */
162
+ export function defaultLibraryCategory(
163
+ type: LibraryType,
164
+ meta: { voice?: string; voiceId?: string },
165
+ ): string | undefined {
166
+ if (type === 'voice') {
167
+ // Check female first: 'female' contains the substring 'male'.
168
+ const probe = `${meta.voiceId ?? ''} ${meta.voice ?? ''}`.toLowerCase()
169
+ if (/female|女/.test(probe)) return 'female'
170
+ if (/male|男/.test(probe)) return 'male'
171
+ return 'custom'
172
+ }
173
+ if (type === 'tts') return sanitizeSegment(meta.voice ?? meta.voiceId ?? 'default')
174
+ return undefined
175
+ }
176
+
177
+ /** Default resource name from the prompt. */
178
+ export function defaultLibraryName(prompt: string): string {
179
+ const flat = prompt.replace(/\s+/g, ' ').trim()
180
+ return flat === '' ? '未命名音频' : (flat.length > LIBRARY_NAME_MAX ? `${flat.slice(0, LIBRARY_NAME_MAX)}…` : flat)
181
+ }
182
+
183
+ async function readLibraryIndex(): Promise<LibraryEntry[]> {
184
+ try {
185
+ const text = await readFile(LIBRARY_INDEX_FILE, 'utf8')
186
+ const parsed = JSON.parse(text) as unknown
187
+ if (!Array.isArray(parsed)) return []
188
+ return parsed.filter(isLibraryEntry)
189
+ } catch {
190
+ return []
191
+ }
192
+ }
193
+
194
+ function isLibraryEntry(value: unknown): value is LibraryEntry {
195
+ if (value === null || typeof value !== 'object') return false
196
+ const raw = value as Record<string, unknown>
197
+ return typeof raw.id === 'string'
198
+ && typeof raw.name === 'string'
199
+ && (raw.type === 'voice' || raw.type === 'music' || raw.type === 'sfx' || raw.type === 'tts')
200
+ && Array.isArray(raw.files)
201
+ && typeof raw.createdAt === 'number'
202
+ && typeof (raw as Record<string, unknown>).provenance === 'object'
203
+ }
204
+
205
+ async function writeLibraryIndex(entries: LibraryEntry[]): Promise<void> {
206
+ await mkdir(LIBRARY_DATA_DIR, { recursive: true })
207
+ await writeFile(LIBRARY_INDEX_FILE, JSON.stringify(entries, null, 2))
208
+ }
209
+
210
+ /** Same-origin URL for a library-relative file path. */
211
+ export function libraryUrlOf(rel: string): string {
212
+ return `${LIBRARY_API.audio}/${rel.split('/').map(segment => encodeURIComponent(segment)).join('/')}`
213
+ }
214
+
215
+ /** Merge-library-entry: copy one audio/ file into library/<type>/<category>/. */
216
+ async function copyIntoLibrary(input: LibraryAudioInput, typeDir: string, category: string): Promise<LibraryFileRef> {
217
+ const stored = await readAudioFile(input.file)
218
+ if (stored === undefined) {
219
+ throw new Error(`音频文件不存在:${input.file}(请重新生成后再入库)`)
220
+ }
221
+ const ext = path.extname(input.file).replace('.', '') || (stored.mime.split('/')[1]?.replace('mpeg', 'mp3') ?? 'bin')
222
+ const rel = `${typeDir}/${category}/${input.id}.${ext}`
223
+ const target = path.join(LIBRARY_DATA_DIR, ...rel.split('/'))
224
+ await mkdir(path.dirname(target), { recursive: true })
225
+ await writeFile(target, stored.data)
226
+ return {
227
+ url: libraryUrlOf(rel),
228
+ rel,
229
+ mime: stored.mime,
230
+ bytes: stored.bytes,
231
+ ...(input.duration === undefined ? {} : { duration: input.duration }),
232
+ ...(input.voiceId === undefined ? {} : { voiceId: input.voiceId }),
233
+ }
234
+ }
235
+
236
+ /**
237
+ * Save one curated library entry: copies the referenced audio files into
238
+ * library/<type>/<category>/ (audio/ files stay untouched) and appends the
239
+ * entry to the index.
240
+ */
241
+ export async function saveToLibrary(input: {
242
+ audioFiles: LibraryAudioInput[]
243
+ type: LibraryType
244
+ category?: string
245
+ name?: string
246
+ tags?: string[]
247
+ note?: string
248
+ provenance: LibraryProvenance
249
+ }): Promise<LibraryEntry> {
250
+ if (input.audioFiles.length === 0) throw new Error('没有可入库的音频文件')
251
+ const typeDir = LIBRARY_TYPE_DIRS[input.type]
252
+ const category = input.category !== undefined && input.category.trim() !== ''
253
+ ? sanitizeSegment(input.category.trim())
254
+ : defaultLibraryCategory(input.type, {
255
+ voice: input.provenance.voice,
256
+ voiceId: input.provenance.voiceId ?? input.audioFiles.find(file => file.voiceId !== undefined)?.voiceId,
257
+ }) ?? 'default'
258
+ const files = await Promise.all(input.audioFiles.map(file => copyIntoLibrary(file, typeDir, category)))
259
+ const rawName = (input.name ?? '').trim()
260
+ const entry: LibraryEntry = {
261
+ id: randomUUID(),
262
+ createdAt: Date.now(),
263
+ type: input.type,
264
+ category: category === 'default' && input.type !== 'voice' && input.type !== 'tts' ? undefined : category,
265
+ name: rawName === '' ? defaultLibraryName(input.provenance.prompt) : rawName,
266
+ tags: Array.isArray(input.tags) ? [...new Set(input.tags.map(tag => tag.trim()).filter(tag => tag !== ''))].slice(0, 20) : [],
267
+ ...(input.note !== undefined && input.note.trim() !== '' ? { note: input.note.trim() } : {}),
268
+ files,
269
+ provenance: input.provenance,
270
+ }
271
+ const entries = await readLibraryIndex()
272
+ entries.unshift(entry)
273
+ await writeLibraryIndex(entries)
274
+ return entry
275
+ }
276
+
277
+ /** Read library entries (newest first). */
278
+ export async function listLibrary(): Promise<LibraryEntry[]> {
279
+ const entries = await readLibraryIndex()
280
+ return [...entries].sort((a, b) => b.createdAt - a.createdAt)
281
+ }
282
+
283
+ /** Move one library-relative file to a new rel path (same volume rename, else copy). */
284
+ async function moveLibraryFile(fromRel: string, toRel: string): Promise<void> {
285
+ const from = path.join(LIBRARY_DATA_DIR, ...fromRel.split('/'))
286
+ const to = path.join(LIBRARY_DATA_DIR, ...toRel.split('/'))
287
+ await mkdir(path.dirname(to), { recursive: true })
288
+ try {
289
+ await rename(from, to)
290
+ } catch {
291
+ const data = await readFile(from)
292
+ await writeFile(to, data)
293
+ await unlink(from)
294
+ }
295
+ }
296
+
297
+ /** Patch name/tags/note/type/category; moving type/category relocates files. */
298
+ export async function updateLibraryEntry(id: string, patch: {
299
+ name?: string
300
+ tags?: string[]
301
+ note?: string
302
+ category?: string
303
+ type?: LibraryType
304
+ }): Promise<LibraryEntry | undefined> {
305
+ const entries = await readLibraryIndex()
306
+ const index = entries.findIndex(entry => entry.id === id)
307
+ if (index < 0) return undefined
308
+ const entry = { ...entries[index]!, files: [...entries[index]!.files] }
309
+ if (patch.type !== undefined && LIBRARY_TYPES_VALID.includes(patch.type)) entry.type = patch.type
310
+ if (patch.name !== undefined) entry.name = patch.name.trim() === '' ? defaultLibraryName(entry.provenance.prompt) : patch.name.trim()
311
+ if (patch.tags !== undefined) entry.tags = [...new Set(patch.tags.map(tag => tag.trim()).filter(tag => tag !== ''))].slice(0, 20)
312
+ if (patch.note !== undefined) entry.note = patch.note.trim() === '' ? undefined : patch.note.trim()
313
+ if (patch.category !== undefined && patch.category.trim() !== '') {
314
+ const next = sanitizeSegment(patch.category.trim())
315
+ if (entry.type === 'voice' || entry.type === 'tts') entry.category = next
316
+ }
317
+ const oldCat = entries[index]!.category ?? 'default'
318
+ const newCat = entry.category ?? 'default'
319
+ if (entries[index]!.type !== entry.type || oldCat !== newCat) {
320
+ const moved: LibraryFileRef[] = []
321
+ for (const file of entry.files) {
322
+ const fileName = file.rel.split('/').pop() ?? ''
323
+ const fromRel = `${LIBRARY_TYPE_DIRS[entries[index]!.type]}/${oldCat}/${fileName}`
324
+ const toRel = `${LIBRARY_TYPE_DIRS[entry.type]}/${newCat}/${fileName}`
325
+ if (fromRel !== toRel) await moveLibraryFile(fromRel, toRel)
326
+ moved.push({ ...file, rel: toRel, url: libraryUrlOf(toRel) })
327
+ }
328
+ entry.files = moved
329
+ }
330
+ entries[index] = entry
331
+ await writeLibraryIndex(entries)
332
+ return entry
333
+ }
334
+
335
+ /** Remove entries and their audio files; best-effort prune empty dirs. */
336
+ export async function removeLibraryEntries(ids: string[]): Promise<LibraryEntry[]> {
337
+ const entries = await readLibraryIndex()
338
+ const doomed = new Set(ids)
339
+ const kept = entries.filter(entry => !doomed.has(entry.id))
340
+ for (const entry of entries) {
341
+ if (!doomed.has(entry.id)) continue
342
+ for (const file of entry.files) {
343
+ try {
344
+ await unlink(path.join(LIBRARY_DATA_DIR, ...file.rel.split('/')))
345
+ } catch {
346
+ // best-effort
347
+ }
348
+ }
349
+ }
350
+ // prune now-empty leaf dirs (best effort, one level under type dirs)
351
+ for (const entry of entries) {
352
+ if (!doomed.has(entry.id)) continue
353
+ try {
354
+ await rmdir(path.dirname(path.join(LIBRARY_DATA_DIR, ...entry.files[0]!.rel.split('/'))), { recursive: false })
355
+ } catch {
356
+ // dir not empty or already gone
357
+ }
358
+ }
359
+ await writeLibraryIndex(kept)
360
+ return kept
361
+ }
362
+
363
+ /** Read one library file by its rel path (whitelisted, traversal-safe). */
364
+ export async function readLibraryFile(rel: string): Promise<{ data: Buffer; mime: string; bytes: number } | undefined> {
365
+ const segments = rel.split('/').filter(segment => segment !== '')
366
+ if (segments.length < 2 || segments.length > 3) return undefined
367
+ const [typeDir, category, fileName] = segments
368
+ if (typeDir === undefined || !Object.values(LIBRARY_TYPE_DIRS).includes(typeDir)) return undefined
369
+ if (category === undefined || sanitizeSegment(category) !== category || category.length > 60) return undefined
370
+ if (fileName === undefined || !/^[0-9a-f-]{36}\.[a-z0-9]{2,5}$/i.test(fileName)) return undefined
371
+ const full = path.join(LIBRARY_DATA_DIR, typeDir, category, fileName)
372
+ if (!full.startsWith(path.join(LIBRARY_DATA_DIR, typeDir, category) + path.sep)) return undefined
373
+ try {
374
+ const data = await readFile(full)
375
+ return { data, mime: mimeFromFile(fileName), bytes: data.byteLength }
376
+ } catch {
377
+ return undefined
378
+ }
379
+ }
380
+
381
+ const LIBRARY_TYPES_VALID = ['voice', 'music', 'sfx', 'tts'] as const