dsh-audiogen 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,9 +26,9 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
26
26
  apiUrl: 'https://api.openai.com/v1',
27
27
  hint: 'OpenAI 官方语音合成接口(/audio/speech)',
28
28
  models: [
29
- { alias: 'tts-1', id: 'tts-1' },
30
- { alias: 'tts-1-hd', id: 'tts-1-hd' },
31
- { alias: 'gpt-4o-mini-tts', id: 'gpt-4o-mini-tts' },
29
+ { alias: 'tts-1', id: 'tts-1', category: 'tts' },
30
+ { alias: 'tts-1-hd', id: 'tts-1-hd', category: 'tts' },
31
+ { alias: 'gpt-4o-mini-tts', id: 'gpt-4o-mini-tts', category: 'tts' },
32
32
  ],
33
33
  },
34
34
  {
@@ -37,22 +37,31 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
37
37
  apiUrl: 'https://api.elevenlabs.io/v1',
38
38
  hint: 'ElevenLabs TTS;模型列表请填写你的 Voice ID(如 Rachel / Adam 等别名)',
39
39
  models: [
40
- { alias: 'Rachel', id: '21m00Tcm4TlvDq8ikWAM' },
41
- { alias: 'Adam', id: 'pNInz6obpgDQGcFmaJgB' },
42
- { alias: 'Antoni', id: 'ErXwobaYiN019PkySvjV' },
43
- { alias: 'Bella', id: 'EXAVITQu4vr4xnSDxMaL' },
40
+ { alias: 'Rachel', id: '21m00Tcm4TlvDq8ikWAM', category: 'tts' },
41
+ { alias: 'Adam', id: 'pNInz6obpgDQGcFmaJgB', category: 'tts' },
42
+ { alias: 'Antoni', id: 'ErXwobaYiN019PkySvjV', category: 'tts' },
43
+ { alias: 'Bella', id: 'EXAVITQu4vr4xnSDxMaL', category: 'tts' },
44
44
  ],
45
45
  },
46
46
  {
47
47
  id: 'minimax',
48
48
  name: 'MiniMax',
49
- apiUrl: 'https://api.minimax.chat/v1',
50
- hint: 'MiniMax 语音合成(T2A);需在 API URL 后按官方要求携带 GroupId 或使用完整接口地址',
49
+ apiUrl: 'https://api.minimaxi.com',
50
+ hint: 'MiniMax 音色设计 / TTS / 音乐生成;可使用“获取可用模型”拉取账号音色',
51
51
  models: [
52
- { alias: 'speech-01-turbo', id: 'speech-01-turbo' },
53
- { alias: 'speech-01-hd', id: 'speech-01-hd' },
54
- { alias: 'speech-02-turbo', id: 'speech-02-turbo' },
55
- { alias: 'speech-02-hd', id: 'speech-02-hd' },
52
+ // TTS models
53
+ { alias: 'speech-2.8-hd', id: 'speech-2.8-hd', category: 'tts' },
54
+ { alias: 'speech-2.8-turbo', id: 'speech-2.8-turbo', category: 'tts' },
55
+ { alias: 'speech-2.6-hd', id: 'speech-2.6-hd', category: 'tts' },
56
+ { alias: 'speech-2.6-turbo', id: 'speech-2.6-turbo', category: 'tts' },
57
+ { alias: 'speech-02-hd', id: 'speech-02-hd', category: 'tts' },
58
+ { alias: 'speech-02-turbo', id: 'speech-02-turbo', category: 'tts' },
59
+ { alias: 'speech-01-hd', id: 'speech-01-hd', category: 'tts' },
60
+ { alias: 'speech-01-turbo', id: 'speech-01-turbo', category: 'tts' },
61
+ // Music models
62
+ { alias: 'music-3.0', id: 'music-3.0', category: 'music' },
63
+ { alias: 'music-2.6', id: 'music-2.6', category: 'music' },
64
+ { alias: 'music-cover', id: 'music-cover', category: 'music' },
56
65
  ],
57
66
  },
58
67
  {
@@ -61,8 +70,8 @@ export const AUDIO_PRESETS: AudioPresetProvider[] = [
61
70
  apiUrl: 'https://api.stability.ai/v2beta/audio',
62
71
  hint: 'Stability AI 音乐/音效生成(stable-audio 系列)',
63
72
  models: [
64
- { alias: 'stable-audio-2.0', id: 'stable-audio-2.0' },
65
- { alias: 'stable-audio-1.0', id: 'stable-audio-1.0' },
73
+ { alias: 'stable-audio-2.0', id: 'stable-audio-2.0', category: 'music' },
74
+ { alias: 'stable-audio-1.0', id: 'stable-audio-1.0', category: 'music' },
66
75
  ],
67
76
  },
68
77
  {
@@ -62,11 +62,15 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
62
62
  const [outputs, setOutputs] = useState<GeneratedAudio[]>([])
63
63
  const { entries, reload, clear } = useHistory()
64
64
 
65
+ const visibleModels = useMemo(() => modelOptions.models
66
+ .filter(entry => entry.category === undefined || entry.category === 'tts' && mode === 'tts' || entry.category === mode)
67
+ .map(entry => entry.alias), [modelOptions.models, mode])
68
+
65
69
  useEffect(() => {
66
- if (modelOptions.models.length > 0 && !modelOptions.models.includes(model)) {
67
- setModel(modelOptions.models[0]!)
70
+ if (visibleModels.length > 0 && !visibleModels.includes(model)) {
71
+ setModel(visibleModels[0]!)
68
72
  }
69
- }, [modelOptions.models, model])
73
+ }, [visibleModels, model])
70
74
 
71
75
  const submit = async (): Promise<void> => {
72
76
  if (prompt.trim() === '') {
@@ -78,7 +82,7 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
78
82
  try {
79
83
  const response = await api.generate({
80
84
  mode,
81
- model: (model || modelOptions.models[0]) ?? '',
85
+ model: (model || visibleModels[0]) ?? '',
82
86
  prompt: prompt.trim(),
83
87
  ...(voice.trim() !== '' ? { voice: voice.trim() } : {}),
84
88
  ...(speed.trim() !== '' ? { speed: Number(speed) } : {}),
@@ -132,8 +136,8 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
132
136
  <label className={css.label}>
133
137
  <span>{tt('model.label')}</span>
134
138
  <select className={css.select} value={model} onChange={event => setModel(event.target.value)}>
135
- {modelOptions.models.length === 0 ? <option value="">(请在设置中添加)</option> : null}
136
- {modelOptions.models.map(item => <option key={item} value={item}>{item}</option>)}
139
+ {visibleModels.length === 0 ? <option value="">(当前模式暂无可用模型)</option> : null}
140
+ {visibleModels.map(item => <option key={item} value={item}>{item}</option>)}
137
141
  </select>
138
142
  </label>
139
143
  {mode === 'tts' ? (
@@ -161,7 +165,7 @@ export function AudioGenPanel(props: { api: AudiogenApi; scope: AudiogenScope })
161
165
  </select>
162
166
  </label>
163
167
  {!connected && <p className={css.hint}>{tt('config.missing')}</p>}
164
- <button type="button" className={css.generate} disabled={loading || !connected} onClick={() => void submit()}>
168
+ <button type="button" className={css.generate} disabled={loading || !connected || visibleModels.length === 0} onClick={() => void submit()}>
165
169
  {loading ? tt('generating') : tt('generate')}
166
170
  </button>
167
171
  </div>
@@ -12,7 +12,7 @@ import { createSnapshotStore, type SnapshotStore } from '@deepseek-ai/dsh-client
12
12
  import { CardForm, booleanField, textField, type CardActions, type CardShell, type FieldState as CardFieldState } from './settings-form.ts'
13
13
  import { ChannelsForm, type ChannelDraft, type ChannelsFormActions, type ChannelsFormState } from './channels-form.ts'
14
14
  import type { AudiogenScope } from './settings-scope.ts'
15
- import { PRESETS_API, type ModelMapping, type PresetProviderView } from '../protocol.ts'
15
+ import { MODEL_API, PRESETS_API, type DiscoveredAudioModel, type ModelMapping, type PresetProviderView } from '../protocol.ts'
16
16
  import type { AudioGenKey } from './locales.ts'
17
17
  import css from './settings-card.module.css'
18
18
 
@@ -98,14 +98,22 @@ function newChannelDraft(preset: PresetProviderView | undefined, existing: Chann
98
98
  }
99
99
 
100
100
  function modelsToText(models: ModelMapping[]): string {
101
- return models.map(model => `${model.alias}=${model.id}`).join('\n')
101
+ return models.map(model => `${model.alias}=${model.id}${model.category === undefined ? '' : ` @${model.category}`}`).join('\n')
102
102
  }
103
103
 
104
104
  function textToModels(text: string): ModelMapping[] {
105
105
  return text.split(/\n|,/).map(line => line.trim()).filter(Boolean).map(line => {
106
- const eq = line.indexOf('=')
107
- if (eq < 0) return { alias: line, id: line }
108
- return { alias: line.slice(0, eq).trim(), id: line.slice(eq + 1).trim() }
106
+ const at = line.lastIndexOf(' @')
107
+ const category = at >= 0 ? line.slice(at + 2).trim() : undefined
108
+ const body = at >= 0 ? line.slice(0, at).trim() : line
109
+ const eq = body.indexOf('=')
110
+ const alias = eq >= 0 ? body.slice(0, eq).trim() : body.trim()
111
+ const id = eq >= 0 ? body.slice(eq + 1).trim() : alias
112
+ return {
113
+ alias,
114
+ id: id === '' ? alias : id,
115
+ ...(category === undefined || category === '' ? {} : { category: category as NonNullable<ModelMapping['category']> }),
116
+ }
109
117
  }).filter(model => model.alias !== '')
110
118
  }
111
119
 
@@ -118,6 +126,8 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
118
126
  const [presets, setPresets] = useState<PresetProviderView[]>([])
119
127
  const [presetError, setPresetError] = useState<string | null>(null)
120
128
  const [confirmDeleteId, setConfirmDeleteId] = useState<string | null>(null)
129
+ const [discovering, setDiscovering] = useState(false)
130
+ const [discoverError, setDiscoverError] = useState<string | null>(null)
121
131
 
122
132
  const [editName, setEditName] = useState('')
123
133
  const [editUrl, setEditUrl] = useState('')
@@ -161,6 +171,37 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
161
171
  setEditingId(null)
162
172
  }
163
173
 
174
+ const discoverModels = async (): Promise<void> => {
175
+ if (editingId === null) return
176
+ setDiscovering(true)
177
+ setDiscoverError(null)
178
+ try {
179
+ const existing = channels.find(channel => channel.id === editingId)
180
+ const response = await fetch(MODEL_API.discover, {
181
+ method: 'POST',
182
+ headers: { 'content-type': 'application/json' },
183
+ body: JSON.stringify({
184
+ channelId: editingId,
185
+ ...(editUrl.trim() !== '' ? { apiUrl: editUrl.trim() } : {}),
186
+ ...(editKey.trim() !== '' ? { apiKey: editKey.trim() } : {}),
187
+ }),
188
+ })
189
+ const body = await response.json() as { ok?: boolean; models?: DiscoveredAudioModel[]; message?: string; source?: string }
190
+ if (body.ok !== true || body.models === undefined) {
191
+ throw new Error(body.message ?? `HTTP ${response.status}`)
192
+ }
193
+ setEditModels(modelsToText([
194
+ ...body.models,
195
+ ...textToModels(editModels).filter(existingModel => !body.models!.some(model => model.id === existingModel.id)),
196
+ ]))
197
+ void existing
198
+ } catch (error) {
199
+ setDiscoverError(error instanceof Error ? error.message : String(error))
200
+ } finally {
201
+ setDiscovering(false)
202
+ }
203
+ }
204
+
164
205
  return (
165
206
  <li className={css.card}>
166
207
  <button
@@ -297,6 +338,12 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
297
338
  <label className={css.label} htmlFor={`audiogen-models-${editing.id}`}>{t('channel.models')}</label>
298
339
  <textarea id={`audiogen-models-${editing.id}`} className={css.textarea} value={editModels} onChange={event => setEditModels(event.target.value)} />
299
340
  <p className={css.sectionHint}>{t('channel.modelsHint')}</p>
341
+ <div className={css.channelAddRow}>
342
+ <button type="button" className={css.channelAdd} disabled={discovering || !state.writable} onClick={() => void discoverModels()}>
343
+ {discovering ? '获取中…' : '获取可用模型'}
344
+ </button>
345
+ </div>
346
+ {discoverError !== null ? <p className={css.failed}>{discoverError}</p> : null}
300
347
  </div>
301
348
  <label className={css.label}>
302
349
  <input type="checkbox" checked={editDefault} onChange={event => setEditDefault(event.target.checked)} /> {t('channel.default')}
@@ -78,7 +78,11 @@ function deepEqualJson(a: unknown, b: unknown): boolean {
78
78
  /** Trim and normalize a draft channel (models never carry empty aliases). */
79
79
  function stripChannel(channel: ChannelDraft): ChannelDraft {
80
80
  const models = channel.models
81
- .map(model => ({ alias: model.alias.trim(), id: model.id.trim() === '' ? model.alias.trim() : model.id.trim() }))
81
+ .map(model => ({
82
+ alias: model.alias.trim(),
83
+ id: model.id.trim() === '' ? model.alias.trim() : model.id.trim(),
84
+ ...(model.category === undefined ? {} : { category: model.category }),
85
+ }))
82
86
  .filter(model => model.alias !== '')
83
87
  return {
84
88
  id: channel.id,
@@ -58,7 +58,7 @@ export const zh = {
58
58
  'channel.apiKey': 'API 密钥',
59
59
  'channel.apiKeyHint': '留空则保持当前密钥;输入新值可更换。',
60
60
  'channel.models': '模型 / 音色(每行一个:别名=上游ID)',
61
- 'channel.modelsHint': '例如 tts-1=tts-1 或 Rachel=21m00Tcm4TlvDq8ikWAM',
61
+ 'channel.modelsHint': '每行格式:别名=上游ID,可加 @分类(tts/music/sfx/voice_design/voice_clone);例如 tts-1=tts-1 @tts 或 Rachel=21m00Tcm4TlvDq8ikWAM @tts',
62
62
  'channel.default': '设为默认',
63
63
  'channel.cancel': '取消',
64
64
  'channel.save': '保存渠道',
@@ -124,7 +124,7 @@ export const en: Record<AudioGenKey, string> = {
124
124
  'channel.apiKey': 'API key',
125
125
  'channel.apiKeyHint': 'Leave blank to keep the current key.',
126
126
  'channel.models': 'Models / voices (one per line: alias=upstreamId)',
127
- 'channel.modelsHint': 'e.g. tts-1=tts-1 or Rachel=21m00Tcm4TlvDq8ikWAM',
127
+ 'channel.modelsHint': 'One per line: alias=upstreamId, optional @category (tts/music/sfx/voice_design/voice_clone); e.g. tts-1=tts-1 @tts or Rachel=21m00Tcm4TlvDq8ikWAM @tts',
128
128
  'channel.default': 'Set default',
129
129
  'channel.cancel': 'Cancel',
130
130
  'channel.save': 'Save channel',
@@ -14,7 +14,7 @@ import {
14
14
  type SettingsScopeSnapshot,
15
15
  type SnapshotStore,
16
16
  } from '@deepseek-ai/dsh-client-runtime/client'
17
- import { SETTINGS_API, type ChannelConfig } from '../protocol.ts'
17
+ import { SETTINGS_API, type AudioModelCategory, type ChannelConfig } from '../protocol.ts'
18
18
 
19
19
  /** The fields this plugin's settings card edits. */
20
20
  export interface AudiogenConfig {
@@ -269,7 +269,10 @@ export function bindAudiogenScope(fetchFn: typeof fetch = fetch): AudiogenScope
269
269
  * Falls back to the legacy flat allow-list while no channels exist (upgrade
270
270
  * path). Pure projection — no host calls.
271
271
  */
272
- export function audioModelOptions(config: AudiogenConfig | undefined): { models: string[]; defaultChannelId?: string } {
272
+ export function audioModelOptions(config: AudiogenConfig | undefined): {
273
+ models: Array<{ alias: string; category?: AudioModelCategory }>
274
+ defaultChannelId?: string
275
+ } {
273
276
  const channels = config?.channels ?? []
274
277
  if (channels.length === 0) {
275
278
  return { models: [] }
@@ -278,11 +281,14 @@ export function audioModelOptions(config: AudiogenConfig | undefined): { models:
278
281
  ? config.defaultChannelId
279
282
  : channels[0]!.id
280
283
  const ordered = [defaultId, ...channels.filter(channel => channel.id !== defaultId).map(channel => channel.id)]
281
- const models: string[] = []
284
+ const models: Array<{ alias: string; category?: AudioModelCategory }> = []
285
+ const seen = new Set<string>()
282
286
  for (const id of ordered) {
283
287
  const channel = channels.find(candidate => candidate.id === id)!
284
288
  for (const model of channel.models) {
285
- if (model.alias !== '' && !models.includes(model.alias)) models.push(model.alias)
289
+ if (model.alias === '' || seen.has(model.alias)) continue
290
+ seen.add(model.alias)
291
+ models.push({ alias: model.alias, ...(model.category === undefined ? {} : { category: model.category }) })
286
292
  }
287
293
  }
288
294
  return models.length > 0 ? { models, defaultChannelId: defaultId } : { models: [], defaultChannelId: defaultId }
package/src/protocol.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
9
9
 
10
10
  /** Published package version shared by the host updater and the client UI. */
11
- export const PLUGIN_VERSION = '0.1.0'
11
+ export const PLUGIN_VERSION = '0.2.0'
12
12
 
13
13
  /** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
14
14
  export const SETTINGS_API = {
@@ -22,6 +22,11 @@ export const GENERATE_API = '/api/dsh-audiogen/generate' as const
22
22
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
23
23
  export const PRESETS_API = '/api/dsh-audiogen/presets' as const
24
24
 
25
+ /** Host-mediated model/voice discovery endpoint. */
26
+ export const MODEL_API = {
27
+ discover: '/api/dsh-audiogen/models/discover',
28
+ } as const
29
+
25
30
  /** Loopback-only audio file reader for panel/tool-result previews. */
26
31
  export const AUDIO_API = {
27
32
  file: '/api/dsh-audiogen/audio',
@@ -42,12 +47,28 @@ export const HISTORY_MAX = 50
42
47
  /** Audio generation modes. */
43
48
  export type AudioMode = 'tts' | 'music' | 'sfx'
44
49
 
50
+ /** The capability category of an audio model/voice. */
51
+ export type AudioModelCategory =
52
+ | 'tts'
53
+ | 'music'
54
+ | 'sfx'
55
+ | 'voice_design'
56
+ | 'voice_clone'
57
+
45
58
  /** One model mapping in a channel's catalog: display alias → upstream id. */
46
59
  export interface ModelMapping {
47
60
  /** User-facing model/voice name (defaults to the upstream id). */
48
61
  alias: string
49
62
  /** Upstream model or voice id sent to the provider. */
50
63
  id: string
64
+ /** Optional capability category, used by the UI to group model lists. */
65
+ category?: AudioModelCategory
66
+ }
67
+
68
+ /** A model/voice discovered from a vendor endpoint (not yet persisted). */
69
+ export interface DiscoveredAudioModel extends ModelMapping {
70
+ /** Optional human-readable description from the vendor. */
71
+ description?: string
51
72
  }
52
73
 
53
74
  /**
package/src/routes.ts CHANGED
@@ -11,10 +11,11 @@ import { randomUUID } from 'node:crypto'
11
11
  import type { WebRoute } from '@deepseek-ai/dsh-host-webserver'
12
12
  import { SettingsConflictError, settingsNamespace, type SettingsDescriptor } from '@deepseek-ai/dsh-settings'
13
13
  import { generateAudio, AudioGenError, type AudioChannel } from './audio-engine.ts'
14
+ import { discoverAudioModels } from './audio-models.ts'
14
15
  import { AUDIO_PRESETS } from './audio-presets.ts'
15
16
  import { appendHistory, clearHistory, listHistory, readAudioFile, removeHistory, saveAudioFile } from './audio-store.ts'
16
17
  import {
17
- AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, PRESETS_API, SETTINGS_API,
18
+ AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, MODEL_API, PRESETS_API, SETTINGS_API,
18
19
  type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput,
19
20
  } from './protocol.ts'
20
21
 
@@ -193,6 +194,32 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
193
194
  writeJson(res, 200, { ok: true, presets: AUDIO_PRESETS })
194
195
  },
195
196
  },
197
+ // ---------------------------------------------- model/voice discovery
198
+ {
199
+ kind: 'exact',
200
+ path: MODEL_API.discover,
201
+ handler: async (req, res) => {
202
+ if (!guard(req, res, 'POST')) return
203
+ const body = await readJsonBody(req)
204
+ const view = deps.resolveChannels()
205
+ const stored = view.channels.find(candidate => candidate.id === (typeof body?.channelId === 'string' ? body.channelId : undefined))
206
+ ?? view.channels.find(candidate => candidate.id === view.defaultChannelId)
207
+ ?? view.channels[0]
208
+ const channel: AudioChannel = {
209
+ id: stored?.id ?? 'preview',
210
+ preset: stored?.preset ?? '',
211
+ name: stored?.name ?? '',
212
+ apiUrl: typeof body?.apiUrl === 'string' && body.apiUrl.trim() !== '' ? body.apiUrl.trim() : (stored?.apiUrl ?? ''),
213
+ apiKey: typeof body?.apiKey === 'string' && body.apiKey.trim() !== '' ? body.apiKey.trim() : (stored?.apiKey ?? ''),
214
+ models: stored?.models ?? [],
215
+ }
216
+ try {
217
+ writeJson(res, 200, { ok: true, ...await discoverAudioModels(channel) })
218
+ } catch (error) {
219
+ writeJson(res, 200, { ok: false, code: 'model-discovery-failed', message: messageOf(error) })
220
+ }
221
+ },
222
+ },
196
223
  // -------------------------------------------------- settings describe
197
224
  {
198
225
  kind: 'exact',