dsh-audiogen 0.4.3 → 0.4.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -101,6 +101,14 @@ function modeLabelOf(mode: AudioMode): string {
101
101
  return tt('mode.voiceDesign')
102
102
  }
103
103
 
104
+ /** 历史时间显示(YYYY-MM-DD HH:mm)。 */
105
+ function formatClock(timestamp: number): string {
106
+ const date = new Date(timestamp)
107
+ if (Number.isNaN(date.getTime())) return ''
108
+ const pad = (n: number): string => String(n).padStart(2, '0')
109
+ return `${date.getFullYear()}-${pad(date.getMonth() + 1)}-${pad(date.getDate())} ${pad(date.getHours())}:${pad(date.getMinutes())}`
110
+ }
111
+
104
112
  function useConfig(scope: AudiogenScope) {
105
113
  const [value, setValue] = useState(scope.getSnapshot().value)
106
114
  useEffect(() => scope.subscribe(() => { setValue(scope.getSnapshot().value) }), [scope])
@@ -222,6 +230,9 @@ export function StudioView(props: {
222
230
  const [bitrate, setBitrate] = useState('')
223
231
  const [audioChannel, setAudioChannel] = useState('')
224
232
  const [subtitle, setSubtitle] = useState(false)
233
+ // 提示词增强
234
+ const [enhancing, setEnhancing] = useState(false)
235
+ const [enhancePreview, setEnhancePreview] = useState<string | null>(null)
225
236
  // Stable Audio 参数(仅 Stability 渠道显示)
226
237
  const [seed, setSeed] = useState('')
227
238
  const [steps, setSteps] = useState('')
@@ -499,6 +510,109 @@ export function StudioView(props: {
499
510
  setTasks(current => current.filter(task => task.id !== taskId))
500
511
  }
501
512
 
513
+ /** 从历史参数回填表单(参考 AI 生图「恢复」):配置 + prompt 一键复用。 */
514
+ const restoreFromParams = (params: Record<string, unknown>, modeValue: AudioMode, singleModel: string, compareModelsRestore?: string[]): void => {
515
+ const str = (key: string): string | undefined => {
516
+ const v = params[key]
517
+ return typeof v === 'string' && v.trim() !== '' ? v.trim() : undefined
518
+ }
519
+ const num = (key: string): number | undefined => {
520
+ const v = params[key]
521
+ if (typeof v === 'number' && Number.isFinite(v)) return v
522
+ if (typeof v === 'string' && v.trim() !== '') {
523
+ const parsed = Number(v)
524
+ return Number.isFinite(parsed) ? parsed : undefined
525
+ }
526
+ return undefined
527
+ }
528
+ const bool = (key: string): boolean | undefined => typeof params[key] === 'boolean' ? params[key] as boolean : undefined
529
+ setMode(modeValue)
530
+ const modelValue = str('model') ?? singleModel
531
+ if (modelValue !== '') setModel(modelValue)
532
+ if (compareModelsRestore !== undefined && compareModelsRestore.length > 0) {
533
+ setCompareMode(true)
534
+ setCompareModels(compareModelsRestore)
535
+ } else {
536
+ setCompareMode(false)
537
+ }
538
+ const voiceValue = str('voice')
539
+ if (voiceValue !== undefined) setVoice(voiceValue)
540
+ const speedValue = num('speed')
541
+ if (speedValue !== undefined) setSpeed(String(speedValue))
542
+ const durationValue = num('duration')
543
+ if (durationValue !== undefined) setDuration(String(durationValue))
544
+ const formatValue = str('format')
545
+ if (formatValue !== undefined) setFormat(formatValue)
546
+ const lyricsValue = str('lyrics')
547
+ if (lyricsValue !== undefined) setLyrics(lyricsValue)
548
+ const instrumentalValue = bool('isInstrumental')
549
+ if (instrumentalValue !== undefined) setInstrumental(instrumentalValue)
550
+ const loopValue = bool('loop')
551
+ if (loopValue !== undefined) setLoop(loopValue)
552
+ const influenceValue = num('promptInfluence')
553
+ if (influenceValue !== undefined) setPromptInfluence(String(influenceValue))
554
+ const emotionValue = str('emotion')
555
+ if (emotionValue !== undefined) setEmotion(emotionValue)
556
+ const volValue = num('vol')
557
+ if (volValue !== undefined) setVol(String(volValue))
558
+ const pitchValue = num('pitch')
559
+ if (pitchValue !== undefined) setPitch(String(pitchValue))
560
+ if (Array.isArray(params.pronunciationTone)) {
561
+ setToneText(params.pronunciationTone.filter((item): item is string => typeof item === 'string').join('\n'))
562
+ }
563
+ const sampleRateValue = num('sampleRate')
564
+ if (sampleRateValue !== undefined) setSampleRate(String(sampleRateValue))
565
+ const bitrateValue = num('bitrate')
566
+ if (bitrateValue !== undefined) setBitrate(String(bitrateValue))
567
+ const channelValue = num('audioChannel')
568
+ if (channelValue !== undefined) setAudioChannel(String(channelValue))
569
+ const subtitleValue = bool('subtitleEnable')
570
+ if (subtitleValue !== undefined) setSubtitle(subtitleValue)
571
+ const seedValue = num('seed')
572
+ if (seedValue !== undefined) setSeed(String(seedValue))
573
+ const stepsValue = num('steps')
574
+ if (stepsValue !== undefined) setSteps(String(stepsValue))
575
+ const cfgValue = num('cfgScale')
576
+ if (cfgValue !== undefined) setCfgScale(String(cfgValue))
577
+ const previewValue = str('previewText')
578
+ if (previewValue !== undefined) setPreviewText(previewValue)
579
+ const channelIdValue = str('channelId')
580
+ if (channelIdValue !== undefined && modeValue === 'voice_design') setDesignChannelId(channelIdValue)
581
+ props.showToast('已恢复该次生成的配置,可直接再次生成')
582
+ }
583
+
584
+ /** 调用宿主增强(Agent 默认模型),结果先预览再应用。 */
585
+ const runEnhance = async (): Promise<void> => {
586
+ if (prompt.trim() === '') {
587
+ setError('请先输入文本/提示词,再点击增强')
588
+ return
589
+ }
590
+ setEnhancing(true)
591
+ setError(null)
592
+ try {
593
+ const result = await api.enhancePrompt(prompt.trim(), mode)
594
+ if (result.ok !== true || result.enhanced === undefined) {
595
+ setError(result.message ?? '增强失败,请稍后重试')
596
+ return
597
+ }
598
+ setEnhancePreview(result.enhanced)
599
+ } catch (err) {
600
+ setError(err instanceof Error ? err.message : String(err))
601
+ } finally {
602
+ setEnhancing(false)
603
+ }
604
+ }
605
+
606
+ /** 删除历史记录(对比任务卡删除该任务的全部模型条目)。 */
607
+ const deleteHistoryEntries = async (ids: string[]): Promise<void> => {
608
+ try {
609
+ for (const id of ids) await api.removeHistory(id)
610
+ } catch {
611
+ // best-effort
612
+ }
613
+ reload()
614
+ }
615
+
502
616
  const openSaveDialog = (files: GeneratedAudio[], context: SaveDialogContext): void => {
503
617
  setSaveDialog({ files, context })
504
618
  }
@@ -767,7 +881,7 @@ export function StudioView(props: {
767
881
  <div className={css.studio}>
768
882
  <div className={css.formCol}>
769
883
  <div className={css.modeRow}>
770
- {(['tts', 'music', 'sfx', 'voice_design'] as const).map(item => (
884
+ {([['tts', '🎙️'], ['music', '🎵'], ['sfx', '🔊'], ['voice_design', '🎨']] as Array<[AudioMode, string]>).map(([item, icon]) => (
771
885
  <button
772
886
  key={item}
773
887
  type="button"
@@ -775,15 +889,35 @@ export function StudioView(props: {
775
889
  data-active={mode === item ? 'true' : 'false'}
776
890
  onClick={() => setMode(item)}
777
891
  >
778
- {item === 'tts' ? tt('mode.tts') : item === 'music' ? tt('mode.music') : item === 'sfx' ? tt('mode.sfx') : tt('mode.voiceDesign')}
892
+ <span className={css.modeIcon}>{icon}</span>
893
+ {modeLabelOf(item)}
779
894
  </button>
780
895
  ))}
781
896
  </div>
782
897
 
898
+ <div className={css.formSectionRow}>
899
+ <p className={css.formSection}>输入</p>
900
+ <button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>
901
+ {enhancing ? '增强中…' : '✨ 增强提示词'}
902
+ </button>
903
+ </div>
783
904
  <label className={css.label}>
784
905
  <span>{mode === 'voice_design' ? '音色描述' : mode === 'tts' ? '文本' : '提示词'}</span>
785
906
  <textarea className={css.textarea} value={prompt} onChange={event => setPrompt(event.target.value)} placeholder={tt('prompt.placeholder')} />
786
907
  </label>
908
+ {enhancePreview !== null ? (
909
+ <div className={css.enhanceCard}>
910
+ <div className={css.enhanceCardHead}>
911
+ <strong>增强结果({modeLabelOf(mode)})</strong>
912
+ <span className={css.enhanceActions}>
913
+ <button type="button" className={css.ghostButton} onClick={() => { setPrompt(enhancePreview); setEnhancePreview(null); props.showToast('已应用增强结果') }}>应用</button>
914
+ <button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>重新生成</button>
915
+ <button type="button" className={css.ghostButton} onClick={() => setEnhancePreview(null)}>放弃</button>
916
+ </span>
917
+ </div>
918
+ <textarea className={css.textarea} value={enhancePreview} readOnly />
919
+ </div>
920
+ ) : null}
787
921
 
788
922
  {mode === 'voice_design' ? (
789
923
  <>
@@ -806,6 +940,7 @@ export function StudioView(props: {
806
940
  </>
807
941
  ) : null}
808
942
 
943
+ <p className={css.formSection}>模型</p>
809
944
  {needModel ? (
810
945
  <label className={css.checkbox} title="选择多个模型,用相同参数逐个生成,便于对比效果">
811
946
  <input type="checkbox" checked={compareMode} onChange={event => setCompareMode(event.target.checked)} />
@@ -902,7 +1037,14 @@ export function StudioView(props: {
902
1037
  )
903
1038
  ) : null}
904
1039
 
905
- {globalSpecs.filter(spec => spec.advanced !== true).map(spec => renderField(spec))}
1040
+ {globalSpecs.some(spec => spec.advanced !== true) ? <p className={css.formSection}>生成参数</p> : null}
1041
+ <div className={css.formFields}>
1042
+ {globalSpecs.filter(spec => spec.advanced !== true).map(spec => (
1043
+ <div key={spec.key} className={spec.key === 'lyrics' || spec.key === 'toneText' ? css.fieldFull : css.fieldCell}>
1044
+ {renderField(spec)}
1045
+ </div>
1046
+ ))}
1047
+ </div>
906
1048
 
907
1049
  {mode === 'tts' && globalSpecs.some(spec => spec.advanced === true) ? (
908
1050
  <details className={css.advanced}>
@@ -935,6 +1077,15 @@ export function StudioView(props: {
935
1077
  <span className={css.resultEmptyIcon}>🎵</span>
936
1078
  <p>{tt('result.empty')}</p>
937
1079
  <p className={css.resultEmptyHint}>点击「开始生成」即创建一个任务,可同时进行多个;勾选「模型对比」用多个模型同参数生成对比</p>
1080
+ <button type="button" className={css.ghostButton} onClick={() => {
1081
+ const examples: Record<AudioMode, string> = {
1082
+ tts: '今天是不是很开心呀(laughs),当然了!我们一起去公园散步吧。',
1083
+ music: 'Cinematic orchestral piece with a clear "before/after" transition at 1:00, starting minimalist piano + strings, then full orchestra entrance with timpani and brass at the 1-minute mark.',
1084
+ sfx: '科技感 UI 提示音:清脆短促,带轻微回声与空气感。',
1085
+ voice_design: '讲述悬疑故事的播音员,声音低沉富有磁性,语速时快时慢,营造紧张神秘的氛围。',
1086
+ }
1087
+ setPrompt(examples[mode] ?? '')
1088
+ }}>填入示例 prompt</button>
938
1089
  </div>
939
1090
  ) : (
940
1091
  <div className={css.taskList}>
@@ -955,6 +1106,9 @@ export function StudioView(props: {
955
1106
  <span className={css.resultModeChip}>{task.mode}</span>
956
1107
  <span className={css.taskLabel} title={task.prompt}>{task.label}</span>
957
1108
  <span className={css.taskStatus} data-state={task.status}>{statusText}</span>
1109
+ {task.status === 'running' ? (
1110
+ <span className={css.taskBar}><i style={{ width: `${task.progress.total > 0 ? Math.round((task.progress.done / task.progress.total) * 100) : 0}%` }} /></span>
1111
+ ) : null}
958
1112
  <span className={css.taskActions}>
959
1113
  {task.status === 'running' ? (
960
1114
  <button type="button" className={css.ghostButton} onClick={() => cancelTask(task.id)}>取消</button>
@@ -1023,6 +1177,7 @@ export function StudioView(props: {
1023
1177
  <summary className={css.historyCompareSummary}>
1024
1178
  <span className={css.historyPrompt}>{item.prompt}</span>
1025
1179
  <span className={css.historyCompareBadge}>对比 · {item.models.length} 个模型</span>
1180
+ <span className={css.historyTime}>{formatClock(item.createdAt)}</span>
1026
1181
  </summary>
1027
1182
  <div className={css.historyMeta}>{modeLabelOf(item.mode)} · {item.models.map(model => model.model).join(' / ')}</div>
1028
1183
  {item.models.map(model => (
@@ -1040,6 +1195,15 @@ export function StudioView(props: {
1040
1195
  </div>
1041
1196
  </div>
1042
1197
  ))}
1198
+ <div className={css.historyActions}>
1199
+ <button type="button" className={css.historyIcon} title="恢复(回填配置与全部模型)" onClick={() => restoreFromParams(
1200
+ (item.models[0]?.entry.params ?? {}) as Record<string, unknown>,
1201
+ item.mode,
1202
+ item.models[0]?.model ?? '',
1203
+ item.models.map(model => model.model),
1204
+ )}>↺</button>
1205
+ <button type="button" className={css.historyIcon} title="删除整个对比任务" onClick={() => void deleteHistoryEntries(item.models.map(model => model.entry.id))}>✕</button>
1206
+ </div>
1043
1207
  </details>
1044
1208
  )
1045
1209
  }
@@ -1048,12 +1212,19 @@ export function StudioView(props: {
1048
1212
  <div className={css.historyItem} key={item.key}>
1049
1213
  <div className={css.historyPrompt}>{entry.prompt}</div>
1050
1214
  <div className={css.historyMeta}>{modeLabelOf(entry.mode)} · {entry.model}{entry.channel ? ` · ${entry.channel}` : ''}</div>
1215
+ <div className={css.historyTime}>{formatClock(entry.createdAt)}</div>
1051
1216
  {entry.audio.map((audio, index) => (
1052
1217
  <AudioPlayer key={index} src={audio.url} compact itemKey={`${entry.id}-${index}`} />
1053
1218
  ))}
1054
1219
  <div className={css.historyActions}>
1055
- <button type="button" className={css.historyAction} onClick={() => openSaveDialog(audioRefsOfEntry(entry), contextOfEntry(entry))}>
1056
- <StarIcon /> 入库
1220
+ <button type="button" className={css.historyIcon} title="恢复(回填配置与 prompt)" onClick={() => restoreFromParams(
1221
+ (entry.params ?? {}) as Record<string, unknown>,
1222
+ entry.mode,
1223
+ entry.model,
1224
+ )}>↺</button>
1225
+ <button type="button" className={css.historyIcon} title="删除这条记录" onClick={() => void deleteHistoryEntries([entry.id])}>✕</button>
1226
+ <button type="button" className={css.historyIcon} title="加入资源库" onClick={() => openSaveDialog(audioRefsOfEntry(entry), contextOfEntry(entry))}>
1227
+ <StarIcon />
1057
1228
  </button>
1058
1229
  </div>
1059
1230
  </div>
package/src/index.ts CHANGED
@@ -16,8 +16,10 @@ import z from 'schemastery'
16
16
  import type {} from '@deepseek-ai/dsh-host-webserver'
17
17
  import type {} from '@deepseek-ai/dsh-system-prompt'
18
18
  import type {} from '@deepseek-ai/dsh-tools'
19
- import { AUDIOGEN_SETTINGS_NAMESPACE, type ChannelConfig, type ModelMapping } from './protocol.ts'
19
+ import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
20
20
  import { createGenerationBudget } from './audio-scheduler.ts'
21
+ import { enhancePromptText } from './prompt-enhance.ts'
22
+ import { AudioGenError } from './audio-engine.ts'
21
23
  import { makeRoutes, type ChannelsView, type SettingsSeam } from './routes.ts'
22
24
  import type { AudioChannel } from './audio-engine.ts'
23
25
  import { registerAgentAudioTools, type AgentAudioToolConfig } from './agent-audio-tools.ts'
@@ -196,6 +198,13 @@ export function apply(ctx: Context, config?: Config): void {
196
198
  // 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
197
199
  const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
198
200
 
201
+ // 提示词增强:复用 Agent 默认模型(agent-default-model 设置),面板与 Agent 工具共用。
202
+ const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
203
+ const seam = ctx.get('settings') as unknown as SettingsSeam
204
+ if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
205
+ return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
206
+ }
207
+
199
208
  const channelsView = (): ChannelsView => {
200
209
  const value = resolve()
201
210
  return { channels: value.channels, defaultChannelId: value.defaultChannelId }
@@ -209,6 +218,7 @@ export function apply(ctx: Context, config?: Config): void {
209
218
  resolveChannels: channelsView,
210
219
  autoSave: () => resolve().autoSaveToLibrary,
211
220
  budget,
221
+ enhance,
212
222
  })
213
223
  const disposers = routes.map(route => ctx.webServer.register(route))
214
224
  return () => { for (const dispose of disposers) dispose() }
@@ -225,6 +235,7 @@ export function apply(ctx: Context, config?: Config): void {
225
235
  defaultChannelId: value.defaultChannelId,
226
236
  autoSaveToLibrary: value.autoSaveToLibrary,
227
237
  budget,
238
+ enhance,
228
239
  }
229
240
  }), 'dsh-audiogen: agent audio tools')
230
241
  })
@@ -0,0 +1,72 @@
1
+ /**
2
+ * 提示词增强:复用 Agent 当前默认模型(设置「模型」里的 provider/model,
3
+ * 即 agent-default-model 命名空间),宿主端发起一次 LLM 调用把用户 prompt
4
+ * 扩展成更适合生成的任务描述。不新增 API key 配置。
5
+ *
6
+ * 面板「✨ 增强提示词」与 generate_audio 工具的 enhance_prompt 都走这里。
7
+ */
8
+
9
+ import type { AudioMode } from './protocol.ts'
10
+ import { AudioGenError } from './audio-engine.ts'
11
+
12
+ /** 按生成模式给出增强指令(系统提示)。 */
13
+ function instructionsFor(mode: AudioMode): string {
14
+ const common = [
15
+ '你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,',
16
+ '请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。',
17
+ '只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。',
18
+ '保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。',
19
+ '描述控制在 200-600 字左右。',
20
+ ].join('')
21
+ const perMode: Record<AudioMode, string> = {
22
+ tts: '这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。',
23
+ music: '这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。',
24
+ sfx: '这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。',
25
+ voice_design: '这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。',
26
+ }
27
+ return common + perMode[mode]
28
+ }
29
+
30
+ export interface PromptEnhanceDeps {
31
+ /** DSH 设置 seam(读 agent-default-model)。 */
32
+ settings: { describe(options?: { redactSecrets?: boolean }): Array<{ ns: unknown; value?: unknown }> }
33
+ /** 宿主 LLM 运行时访问器(延迟读取,调用时才获取)。 */
34
+ llm?: () => unknown
35
+ }
36
+
37
+ /** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
38
+ export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string, mode: AudioMode): Promise<string> {
39
+ const text = prompt.trim()
40
+ if (text === '') throw new AudioGenError('提示词为空,无法增强', 'enhance-empty-prompt')
41
+ const descriptor = (deps.settings.describe({ redactSecrets: true }) ?? []).find(candidate => String(candidate.ns) === 'agent-default-model')
42
+ const value = (descriptor?.value ?? {}) as { provider?: unknown; model?: unknown }
43
+ const provider = typeof value.provider === 'string' && value.provider.trim() !== '' ? value.provider.trim() : ''
44
+ const model = typeof value.model === 'string' && value.model.trim() !== '' ? value.model.trim() : ''
45
+ if (provider === '' || model === '') {
46
+ throw new AudioGenError('未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型', 'no-default-model')
47
+ }
48
+ const runtime = deps.llm?.() as { stream?: (options: unknown) => AsyncIterable<unknown> } | undefined
49
+ if (runtime === undefined || runtime.stream === undefined) {
50
+ throw new AudioGenError('宿主 LLM 服务不可用(ctx.llm 未注册)', 'llm-unavailable')
51
+ }
52
+ let output = ''
53
+ for await (const chunk of runtime.stream({
54
+ provider,
55
+ model,
56
+ messages: [{ role: 'user', content: text }],
57
+ system: instructionsFor(mode),
58
+ temperature: 0.7,
59
+ maxTokens: 1200,
60
+ })) {
61
+ const record = chunk as { type?: string; text?: string }
62
+ if (record.type === 'text-delta' && typeof record.text === 'string') output += record.text
63
+ }
64
+ return stripFences(output.trim())
65
+ }
66
+
67
+ /** 去掉模型可能包裹的 ``` 代码围栏。 */
68
+ function stripFences(value: string): string {
69
+ if (value === '') return value
70
+ const withoutFence = value.replace(/^```[a-zA-Z]*\s*\n?/, '').replace(/\n?```\s*$/, '')
71
+ return withoutFence.trim()
72
+ }
package/src/protocol.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
9
9
 
10
10
  /** Published package version shared by the host updater and the client UI. */
11
- export const PLUGIN_VERSION = '0.4.3'
11
+ export const PLUGIN_VERSION = '0.4.5'
12
12
 
13
13
  /** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
14
14
  export const SETTINGS_API = {
@@ -24,6 +24,9 @@ export const TASK_API = {
24
24
  cancel: '/api/dsh-audiogen/task/cancel',
25
25
  } as const
26
26
 
27
+ /** Loopback-only prompt enhancement route (uses the agent's default model). */
28
+ export const ENHANCE_API = '/api/dsh-audiogen/prompt/enhance' as const
29
+
27
30
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
28
31
  export const PRESETS_API = '/api/dsh-audiogen/presets' as const
29
32
 
package/src/routes.ts CHANGED
@@ -16,7 +16,7 @@ import { discoverAudioModels } from './audio-models.ts'
16
16
  import { AUDIO_PRESETS } from './audio-presets.ts'
17
17
  import { appendHistory, clearHistory, listHistory, readAudioFile, removeHistory, saveAudioFile, listLibrary, saveToLibrary, updateLibraryEntry, removeLibraryEntries, readLibraryFile } from './audio-store.ts'
18
18
  import {
19
- AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
19
+ AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
20
20
  LIBRARY_TYPES,
21
21
  type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput, type LibraryAudioInput, type LibraryProvenance, type LibraryType,
22
22
  } from './protocol.ts'
@@ -47,6 +47,8 @@ export interface AudiogenRoutesDeps {
47
47
  autoSave: () => boolean
48
48
  /** Global upstream concurrency gate (maxConcurrentGenerations). */
49
49
  budget: GenerationBudget
50
+ /** 提示词增强:调用 Agent 默认模型,返回增强后的文本。 */
51
+ enhance: (prompt: string, mode: GenerateAudioRequest['mode']) => Promise<string>
50
52
  }
51
53
 
52
54
  function isLoopbackRequest(request: IncomingMessage): boolean {
@@ -528,6 +530,27 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
528
530
  writeJson(res, 200, { ok: true, aborted: controllers !== undefined ? controllers.size : 0 })
529
531
  },
530
532
  },
533
+ // ------------------------------------------------------- prompt enhance
534
+ {
535
+ kind: 'exact',
536
+ path: ENHANCE_API,
537
+ handler: async (req, res) => {
538
+ if (!guard(req, res, 'POST')) return
539
+ const body = await readJsonBody(req)
540
+ const prompt = typeof body?.prompt === 'string' ? body.prompt.trim() : ''
541
+ if (prompt === '') {
542
+ writeJson(res, 200, { ok: false, code: 'bad-request', message: 'prompt is required' })
543
+ return
544
+ }
545
+ const mode = body?.mode === 'music' ? 'music' : body?.mode === 'sfx' ? 'sfx' : body?.mode === 'voice_design' ? 'voice_design' : 'tts'
546
+ try {
547
+ const enhanced = await deps.enhance(prompt, mode)
548
+ writeJson(res, 200, { ok: true, enhanced })
549
+ } catch (error) {
550
+ writeJson(res, 200, { ok: false, code: 'enhance-failed', message: messageOf(error) })
551
+ }
552
+ },
553
+ },
531
554
  // ----------------------------------------------------------- audio file
532
555
  {
533
556
  kind: 'prefix',