dsh-audiogen 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,7 +12,7 @@ import { createSnapshotStore, type SnapshotStore } from '@deepseek-ai/dsh-client
12
12
  import { CardForm, booleanField, textField, type CardActions, type CardShell, type FieldState as CardFieldState } from './settings-form.ts'
13
13
  import { ChannelsForm, type ChannelDraft, type ChannelsFormActions, type ChannelsFormState } from './channels-form.ts'
14
14
  import type { AudiogenScope } from './settings-scope.ts'
15
- import { PRESETS_API, type ModelMapping, type PresetProviderView } from '../protocol.ts'
15
+ import { MODEL_API, PRESETS_API, type DiscoveredAudioModel, type ModelMapping, type PresetProviderView } from '../protocol.ts'
16
16
  import type { AudioGenKey } from './locales.ts'
17
17
  import css from './settings-card.module.css'
18
18
 
@@ -98,14 +98,22 @@ function newChannelDraft(preset: PresetProviderView | undefined, existing: Chann
98
98
  }
99
99
 
100
100
  function modelsToText(models: ModelMapping[]): string {
101
- return models.map(model => `${model.alias}=${model.id}`).join('\n')
101
+ return models.map(model => `${model.alias}=${model.id}${model.category === undefined ? '' : ` @${model.category}`}`).join('\n')
102
102
  }
103
103
 
104
104
  function textToModels(text: string): ModelMapping[] {
105
105
  return text.split(/\n|,/).map(line => line.trim()).filter(Boolean).map(line => {
106
- const eq = line.indexOf('=')
107
- if (eq < 0) return { alias: line, id: line }
108
- return { alias: line.slice(0, eq).trim(), id: line.slice(eq + 1).trim() }
106
+ const at = line.lastIndexOf(' @')
107
+ const category = at >= 0 ? line.slice(at + 2).trim() : undefined
108
+ const body = at >= 0 ? line.slice(0, at).trim() : line
109
+ const eq = body.indexOf('=')
110
+ const alias = eq >= 0 ? body.slice(0, eq).trim() : body.trim()
111
+ const id = eq >= 0 ? body.slice(eq + 1).trim() : alias
112
+ return {
113
+ alias,
114
+ id: id === '' ? alias : id,
115
+ ...(category === undefined || category === '' ? {} : { category: category as NonNullable<ModelMapping['category']> }),
116
+ }
109
117
  }).filter(model => model.alias !== '')
110
118
  }
111
119
 
@@ -118,6 +126,8 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
118
126
  const [presets, setPresets] = useState<PresetProviderView[]>([])
119
127
  const [presetError, setPresetError] = useState<string | null>(null)
120
128
  const [confirmDeleteId, setConfirmDeleteId] = useState<string | null>(null)
129
+ const [discovering, setDiscovering] = useState(false)
130
+ const [discoverError, setDiscoverError] = useState<string | null>(null)
121
131
 
122
132
  const [editName, setEditName] = useState('')
123
133
  const [editUrl, setEditUrl] = useState('')
@@ -161,6 +171,37 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
161
171
  setEditingId(null)
162
172
  }
163
173
 
174
+ const discoverModels = async (): Promise<void> => {
175
+ if (editingId === null) return
176
+ setDiscovering(true)
177
+ setDiscoverError(null)
178
+ try {
179
+ const existing = channels.find(channel => channel.id === editingId)
180
+ const response = await fetch(MODEL_API.discover, {
181
+ method: 'POST',
182
+ headers: { 'content-type': 'application/json' },
183
+ body: JSON.stringify({
184
+ channelId: editingId,
185
+ ...(editUrl.trim() !== '' ? { apiUrl: editUrl.trim() } : {}),
186
+ ...(editKey.trim() !== '' ? { apiKey: editKey.trim() } : {}),
187
+ }),
188
+ })
189
+ const body = await response.json() as { ok?: boolean; models?: DiscoveredAudioModel[]; message?: string; source?: string }
190
+ if (body.ok !== true || body.models === undefined) {
191
+ throw new Error(body.message ?? `HTTP ${response.status}`)
192
+ }
193
+ setEditModels(modelsToText([
194
+ ...body.models,
195
+ ...textToModels(editModels).filter(existingModel => !body.models!.some(model => model.id === existingModel.id)),
196
+ ]))
197
+ void existing
198
+ } catch (error) {
199
+ setDiscoverError(error instanceof Error ? error.message : String(error))
200
+ } finally {
201
+ setDiscovering(false)
202
+ }
203
+ }
204
+
164
205
  return (
165
206
  <li className={css.card}>
166
207
  <button
@@ -297,6 +338,12 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
297
338
  <label className={css.label} htmlFor={`audiogen-models-${editing.id}`}>{t('channel.models')}</label>
298
339
  <textarea id={`audiogen-models-${editing.id}`} className={css.textarea} value={editModels} onChange={event => setEditModels(event.target.value)} />
299
340
  <p className={css.sectionHint}>{t('channel.modelsHint')}</p>
341
+ <div className={css.channelAddRow}>
342
+ <button type="button" className={css.channelAdd} disabled={discovering || !state.writable} onClick={() => void discoverModels()}>
343
+ {discovering ? '获取中…' : '获取可用模型'}
344
+ </button>
345
+ </div>
346
+ {discoverError !== null ? <p className={css.failed}>{discoverError}</p> : null}
300
347
  </div>
301
348
  <label className={css.label}>
302
349
  <input type="checkbox" checked={editDefault} onChange={event => setEditDefault(event.target.checked)} /> {t('channel.default')}
@@ -78,7 +78,11 @@ function deepEqualJson(a: unknown, b: unknown): boolean {
78
78
  /** Trim and normalize a draft channel (models never carry empty aliases). */
79
79
  function stripChannel(channel: ChannelDraft): ChannelDraft {
80
80
  const models = channel.models
81
- .map(model => ({ alias: model.alias.trim(), id: model.id.trim() === '' ? model.alias.trim() : model.id.trim() }))
81
+ .map(model => ({
82
+ alias: model.alias.trim(),
83
+ id: model.id.trim() === '' ? model.alias.trim() : model.id.trim(),
84
+ ...(model.category === undefined ? {} : { category: model.category }),
85
+ }))
82
86
  .filter(model => model.alias !== '')
83
87
  return {
84
88
  id: channel.id,
@@ -9,7 +9,8 @@ export const zh = {
9
9
  'mode.tts': '文本转语音',
10
10
  'mode.music': '音乐生成',
11
11
  'mode.sfx': '音效生成',
12
- 'prompt.placeholder': '输入要朗读的文本,或描述想生成的音乐 / 音效…',
12
+ 'mode.voiceDesign': '音色设计',
13
+ 'prompt.placeholder': '输入文本、音乐/音效描述,或音色设计描述…',
13
14
  'prompt.required': '请输入文本或提示词',
14
15
  'model.label': '模型 / 音色',
15
16
  'voice.label': '音色',
@@ -58,7 +59,7 @@ export const zh = {
58
59
  'channel.apiKey': 'API 密钥',
59
60
  'channel.apiKeyHint': '留空则保持当前密钥;输入新值可更换。',
60
61
  'channel.models': '模型 / 音色(每行一个:别名=上游ID)',
61
- 'channel.modelsHint': '例如 tts-1=tts-1 或 Rachel=21m00Tcm4TlvDq8ikWAM',
62
+ 'channel.modelsHint': '每行格式:别名=上游ID,可加 @分类(tts/music/sfx/voice_design/voice_clone);例如 tts-1=tts-1 @tts 或 Rachel=21m00Tcm4TlvDq8ikWAM @tts',
62
63
  'channel.default': '设为默认',
63
64
  'channel.cancel': '取消',
64
65
  'channel.save': '保存渠道',
@@ -75,7 +76,8 @@ export const en: Record<AudioGenKey, string> = {
75
76
  'mode.tts': 'Text to speech',
76
77
  'mode.music': 'Music',
77
78
  'mode.sfx': 'Sound effects',
78
- 'prompt.placeholder': 'Text to speak, or a description of the music / sound effect…',
79
+ 'mode.voiceDesign': 'Voice design',
80
+ 'prompt.placeholder': 'Text to speak, or a music/SFX/voice-design description…',
79
81
  'prompt.required': 'Prompt or text is required',
80
82
  'model.label': 'Model / voice',
81
83
  'voice.label': 'Voice',
@@ -124,7 +126,7 @@ export const en: Record<AudioGenKey, string> = {
124
126
  'channel.apiKey': 'API key',
125
127
  'channel.apiKeyHint': 'Leave blank to keep the current key.',
126
128
  'channel.models': 'Models / voices (one per line: alias=upstreamId)',
127
- 'channel.modelsHint': 'e.g. tts-1=tts-1 or Rachel=21m00Tcm4TlvDq8ikWAM',
129
+ 'channel.modelsHint': 'One per line: alias=upstreamId, optional @category (tts/music/sfx/voice_design/voice_clone); e.g. tts-1=tts-1 @tts or Rachel=21m00Tcm4TlvDq8ikWAM @tts',
128
130
  'channel.default': 'Set default',
129
131
  'channel.cancel': 'Cancel',
130
132
  'channel.save': 'Save channel',
@@ -14,7 +14,7 @@ import {
14
14
  type SettingsScopeSnapshot,
15
15
  type SnapshotStore,
16
16
  } from '@deepseek-ai/dsh-client-runtime/client'
17
- import { SETTINGS_API, type ChannelConfig } from '../protocol.ts'
17
+ import { SETTINGS_API, type AudioModelCategory, type ChannelConfig } from '../protocol.ts'
18
18
 
19
19
  /** The fields this plugin's settings card edits. */
20
20
  export interface AudiogenConfig {
@@ -269,7 +269,10 @@ export function bindAudiogenScope(fetchFn: typeof fetch = fetch): AudiogenScope
269
269
  * Falls back to the legacy flat allow-list while no channels exist (upgrade
270
270
  * path). Pure projection — no host calls.
271
271
  */
272
- export function audioModelOptions(config: AudiogenConfig | undefined): { models: string[]; defaultChannelId?: string } {
272
+ export function audioModelOptions(config: AudiogenConfig | undefined): {
273
+ models: Array<{ alias: string; category?: AudioModelCategory }>
274
+ defaultChannelId?: string
275
+ } {
273
276
  const channels = config?.channels ?? []
274
277
  if (channels.length === 0) {
275
278
  return { models: [] }
@@ -278,11 +281,14 @@ export function audioModelOptions(config: AudiogenConfig | undefined): { models:
278
281
  ? config.defaultChannelId
279
282
  : channels[0]!.id
280
283
  const ordered = [defaultId, ...channels.filter(channel => channel.id !== defaultId).map(channel => channel.id)]
281
- const models: string[] = []
284
+ const models: Array<{ alias: string; category?: AudioModelCategory }> = []
285
+ const seen = new Set<string>()
282
286
  for (const id of ordered) {
283
287
  const channel = channels.find(candidate => candidate.id === id)!
284
288
  for (const model of channel.models) {
285
- if (model.alias !== '' && !models.includes(model.alias)) models.push(model.alias)
289
+ if (model.alias === '' || seen.has(model.alias)) continue
290
+ seen.add(model.alias)
291
+ models.push({ alias: model.alias, ...(model.category === undefined ? {} : { category: model.category }) })
286
292
  }
287
293
  }
288
294
  return models.length > 0 ? { models, defaultChannelId: defaultId } : { models: [], defaultChannelId: defaultId }
package/src/protocol.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
9
9
 
10
10
  /** Published package version shared by the host updater and the client UI. */
11
- export const PLUGIN_VERSION = '0.1.0'
11
+ export const PLUGIN_VERSION = '0.3.0'
12
12
 
13
13
  /** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
14
14
  export const SETTINGS_API = {
@@ -22,6 +22,11 @@ export const GENERATE_API = '/api/dsh-audiogen/generate' as const
22
22
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
23
23
  export const PRESETS_API = '/api/dsh-audiogen/presets' as const
24
24
 
25
+ /** Host-mediated model/voice discovery endpoint. */
26
+ export const MODEL_API = {
27
+ discover: '/api/dsh-audiogen/models/discover',
28
+ } as const
29
+
25
30
  /** Loopback-only audio file reader for panel/tool-result previews. */
26
31
  export const AUDIO_API = {
27
32
  file: '/api/dsh-audiogen/audio',
@@ -40,7 +45,15 @@ export const HISTORY_API = {
40
45
  export const HISTORY_MAX = 50
41
46
 
42
47
  /** Audio generation modes. */
43
- export type AudioMode = 'tts' | 'music' | 'sfx'
48
+ export type AudioMode = 'tts' | 'music' | 'sfx' | 'voice_design'
49
+
50
+ /** The capability category of an audio model/voice. */
51
+ export type AudioModelCategory =
52
+ | 'tts'
53
+ | 'music'
54
+ | 'sfx'
55
+ | 'voice_design'
56
+ | 'voice_clone'
44
57
 
45
58
  /** One model mapping in a channel's catalog: display alias → upstream id. */
46
59
  export interface ModelMapping {
@@ -48,6 +61,14 @@ export interface ModelMapping {
48
61
  alias: string
49
62
  /** Upstream model or voice id sent to the provider. */
50
63
  id: string
64
+ /** Optional capability category, used by the UI to group model lists. */
65
+ category?: AudioModelCategory
66
+ }
67
+
68
+ /** A model/voice discovered from a vendor endpoint (not yet persisted). */
69
+ export interface DiscoveredAudioModel extends ModelMapping {
70
+ /** Optional human-readable description from the vendor. */
71
+ description?: string
51
72
  }
52
73
 
53
74
  /**
@@ -85,6 +106,8 @@ export interface GenerateAudioRequest {
85
106
  prompt: string
86
107
  /** Optional voice alias for TTS. */
87
108
  voice?: string
109
+ /** Optional preview text for voice-design APIs. */
110
+ previewText?: string
88
111
  /** Optional speaking rate / speed multiplier. */
89
112
  speed?: number
90
113
  /** Requested duration in seconds (music/sfx). */
@@ -113,6 +136,8 @@ export interface GeneratedAudio {
113
136
  url: string
114
137
  /** Stable audio id / file name. */
115
138
  id: string
139
+ /** Optional voice id returned by a voice-design API. */
140
+ voiceId?: string
116
141
  }
117
142
 
118
143
  /** Successful generate outcome. */
@@ -132,6 +157,8 @@ export interface HistoryAudioRef {
132
157
  mime: string
133
158
  /** Duration in seconds when known. */
134
159
  duration?: number
160
+ /** Optional generated voice id. */
161
+ voiceId?: string
135
162
  }
136
163
 
137
164
  /** A saved generation as the browser consumes it. */
@@ -142,6 +169,7 @@ export interface HistoryEntry {
142
169
  model: string
143
170
  prompt: string
144
171
  voice?: string
172
+ voiceId?: string
145
173
  speed?: number
146
174
  duration?: number
147
175
  format?: string
@@ -158,6 +186,7 @@ export interface HistoryEntryInput {
158
186
  model: string
159
187
  prompt: string
160
188
  voice?: string
189
+ voiceId?: string
161
190
  speed?: number
162
191
  duration?: number
163
192
  format?: string
package/src/routes.ts CHANGED
@@ -11,10 +11,11 @@ import { randomUUID } from 'node:crypto'
11
11
  import type { WebRoute } from '@deepseek-ai/dsh-host-webserver'
12
12
  import { SettingsConflictError, settingsNamespace, type SettingsDescriptor } from '@deepseek-ai/dsh-settings'
13
13
  import { generateAudio, AudioGenError, type AudioChannel } from './audio-engine.ts'
14
+ import { discoverAudioModels } from './audio-models.ts'
14
15
  import { AUDIO_PRESETS } from './audio-presets.ts'
15
16
  import { appendHistory, clearHistory, listHistory, readAudioFile, removeHistory, saveAudioFile } from './audio-store.ts'
16
17
  import {
17
- AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, PRESETS_API, SETTINGS_API,
18
+ AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, MODEL_API, PRESETS_API, SETTINGS_API,
18
19
  type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput,
19
20
  } from './protocol.ts'
20
21
 
@@ -89,7 +90,7 @@ function messageOf(error: unknown): string {
89
90
  }
90
91
 
91
92
  function parseGenerateRequest(body: Record<string, unknown>): GenerateAudioRequest | undefined {
92
- const mode = body.mode === 'music' ? 'music' : body.mode === 'sfx' ? 'sfx' : 'tts'
93
+ const mode = body.mode === 'music' ? 'music' : body.mode === 'sfx' ? 'sfx' : body.mode === 'voice_design' ? 'voice_design' : 'tts'
93
94
  const prompt = typeof body.prompt === 'string' ? body.prompt.trim() : ''
94
95
  if (prompt === '') return undefined
95
96
  return {
@@ -97,6 +98,7 @@ function parseGenerateRequest(body: Record<string, unknown>): GenerateAudioReque
97
98
  model: typeof body.model === 'string' ? body.model.trim() : '',
98
99
  prompt,
99
100
  ...(typeof body.voice === 'string' && body.voice.trim() !== '' ? { voice: body.voice.trim() } : {}),
101
+ ...(typeof body.previewText === 'string' && body.previewText.trim() !== '' ? { previewText: body.previewText.trim() } : {}),
100
102
  ...(typeof body.speed === 'number' ? { speed: body.speed } : {}),
101
103
  ...(typeof body.duration === 'number' ? { duration: body.duration } : {}),
102
104
  ...(typeof body.format === 'string' && body.format.trim() !== '' ? { format: body.format.trim() } : {}),
@@ -140,6 +142,10 @@ function resolveChannelRequest(
140
142
  const target = explicit ?? defaults
141
143
  const asked = request.model.trim()
142
144
  if (asked === '') {
145
+ if (request.mode === 'voice_design') {
146
+ if (target === undefined) return { ok: false, code: 'no-channels', message: '尚未配置任何渠道' }
147
+ return { ok: true, request: { ...request, channelId: target.id, channel: target.name } }
148
+ }
143
149
  const alias = target?.models[0]?.alias ?? ''
144
150
  if (alias === '') {
145
151
  return { ok: false, code: 'no-models', message: `渠道「${target?.name ?? ''}」尚未配置模型/音色,请先在设置中添加` }
@@ -193,6 +199,32 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
193
199
  writeJson(res, 200, { ok: true, presets: AUDIO_PRESETS })
194
200
  },
195
201
  },
202
+ // ---------------------------------------------- model/voice discovery
203
+ {
204
+ kind: 'exact',
205
+ path: MODEL_API.discover,
206
+ handler: async (req, res) => {
207
+ if (!guard(req, res, 'POST')) return
208
+ const body = await readJsonBody(req)
209
+ const view = deps.resolveChannels()
210
+ const stored = view.channels.find(candidate => candidate.id === (typeof body?.channelId === 'string' ? body.channelId : undefined))
211
+ ?? view.channels.find(candidate => candidate.id === view.defaultChannelId)
212
+ ?? view.channels[0]
213
+ const channel: AudioChannel = {
214
+ id: stored?.id ?? 'preview',
215
+ preset: stored?.preset ?? '',
216
+ name: stored?.name ?? '',
217
+ apiUrl: typeof body?.apiUrl === 'string' && body.apiUrl.trim() !== '' ? body.apiUrl.trim() : (stored?.apiUrl ?? ''),
218
+ apiKey: typeof body?.apiKey === 'string' && body.apiKey.trim() !== '' ? body.apiKey.trim() : (stored?.apiKey ?? ''),
219
+ models: stored?.models ?? [],
220
+ }
221
+ try {
222
+ writeJson(res, 200, { ok: true, ...await discoverAudioModels(channel) })
223
+ } catch (error) {
224
+ writeJson(res, 200, { ok: false, code: 'model-discovery-failed', message: messageOf(error) })
225
+ }
226
+ },
227
+ },
196
228
  // -------------------------------------------------- settings describe
197
229
  {
198
230
  kind: 'exact',
@@ -273,6 +305,7 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
273
305
  mime: saved.mime,
274
306
  bytes: saved.bytes,
275
307
  url: `${AUDIO_API.file}/${encodeURIComponent(saved.file)}`,
308
+ ...(output.voiceId === undefined ? {} : { voiceId: output.voiceId }),
276
309
  })
277
310
  }
278
311
  let history