dsh-audiogen 0.4.3 → 0.4.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +623 -355
- package/lib/client.js.map +1 -1
- package/lib/index.js +100 -2
- package/package.json +1 -1
- package/src/agent-audio-tools.ts +11 -0
- package/src/client/api.ts +15 -1
- package/src/client/audio-panel.module.css +193 -12
- package/src/client/studio-view.tsx +176 -5
- package/src/index.ts +12 -1
- package/src/prompt-enhance.ts +72 -0
- package/src/protocol.ts +4 -1
- package/src/routes.ts +24 -1
|
@@ -101,6 +101,14 @@ function modeLabelOf(mode: AudioMode): string {
|
|
|
101
101
|
return tt('mode.voiceDesign')
|
|
102
102
|
}
|
|
103
103
|
|
|
104
|
+
/** 历史时间显示(YYYY-MM-DD HH:mm)。 */
|
|
105
|
+
function formatClock(timestamp: number): string {
|
|
106
|
+
const date = new Date(timestamp)
|
|
107
|
+
if (Number.isNaN(date.getTime())) return ''
|
|
108
|
+
const pad = (n: number): string => String(n).padStart(2, '0')
|
|
109
|
+
return `${date.getFullYear()}-${pad(date.getMonth() + 1)}-${pad(date.getDate())} ${pad(date.getHours())}:${pad(date.getMinutes())}`
|
|
110
|
+
}
|
|
111
|
+
|
|
104
112
|
function useConfig(scope: AudiogenScope) {
|
|
105
113
|
const [value, setValue] = useState(scope.getSnapshot().value)
|
|
106
114
|
useEffect(() => scope.subscribe(() => { setValue(scope.getSnapshot().value) }), [scope])
|
|
@@ -222,6 +230,9 @@ export function StudioView(props: {
|
|
|
222
230
|
const [bitrate, setBitrate] = useState('')
|
|
223
231
|
const [audioChannel, setAudioChannel] = useState('')
|
|
224
232
|
const [subtitle, setSubtitle] = useState(false)
|
|
233
|
+
// 提示词增强
|
|
234
|
+
const [enhancing, setEnhancing] = useState(false)
|
|
235
|
+
const [enhancePreview, setEnhancePreview] = useState<string | null>(null)
|
|
225
236
|
// Stable Audio 参数(仅 Stability 渠道显示)
|
|
226
237
|
const [seed, setSeed] = useState('')
|
|
227
238
|
const [steps, setSteps] = useState('')
|
|
@@ -499,6 +510,109 @@ export function StudioView(props: {
|
|
|
499
510
|
setTasks(current => current.filter(task => task.id !== taskId))
|
|
500
511
|
}
|
|
501
512
|
|
|
513
|
+
/** 从历史参数回填表单(参考 AI 生图「恢复」):配置 + prompt 一键复用。 */
|
|
514
|
+
const restoreFromParams = (params: Record<string, unknown>, modeValue: AudioMode, singleModel: string, compareModelsRestore?: string[]): void => {
|
|
515
|
+
const str = (key: string): string | undefined => {
|
|
516
|
+
const v = params[key]
|
|
517
|
+
return typeof v === 'string' && v.trim() !== '' ? v.trim() : undefined
|
|
518
|
+
}
|
|
519
|
+
const num = (key: string): number | undefined => {
|
|
520
|
+
const v = params[key]
|
|
521
|
+
if (typeof v === 'number' && Number.isFinite(v)) return v
|
|
522
|
+
if (typeof v === 'string' && v.trim() !== '') {
|
|
523
|
+
const parsed = Number(v)
|
|
524
|
+
return Number.isFinite(parsed) ? parsed : undefined
|
|
525
|
+
}
|
|
526
|
+
return undefined
|
|
527
|
+
}
|
|
528
|
+
const bool = (key: string): boolean | undefined => typeof params[key] === 'boolean' ? params[key] as boolean : undefined
|
|
529
|
+
setMode(modeValue)
|
|
530
|
+
const modelValue = str('model') ?? singleModel
|
|
531
|
+
if (modelValue !== '') setModel(modelValue)
|
|
532
|
+
if (compareModelsRestore !== undefined && compareModelsRestore.length > 0) {
|
|
533
|
+
setCompareMode(true)
|
|
534
|
+
setCompareModels(compareModelsRestore)
|
|
535
|
+
} else {
|
|
536
|
+
setCompareMode(false)
|
|
537
|
+
}
|
|
538
|
+
const voiceValue = str('voice')
|
|
539
|
+
if (voiceValue !== undefined) setVoice(voiceValue)
|
|
540
|
+
const speedValue = num('speed')
|
|
541
|
+
if (speedValue !== undefined) setSpeed(String(speedValue))
|
|
542
|
+
const durationValue = num('duration')
|
|
543
|
+
if (durationValue !== undefined) setDuration(String(durationValue))
|
|
544
|
+
const formatValue = str('format')
|
|
545
|
+
if (formatValue !== undefined) setFormat(formatValue)
|
|
546
|
+
const lyricsValue = str('lyrics')
|
|
547
|
+
if (lyricsValue !== undefined) setLyrics(lyricsValue)
|
|
548
|
+
const instrumentalValue = bool('isInstrumental')
|
|
549
|
+
if (instrumentalValue !== undefined) setInstrumental(instrumentalValue)
|
|
550
|
+
const loopValue = bool('loop')
|
|
551
|
+
if (loopValue !== undefined) setLoop(loopValue)
|
|
552
|
+
const influenceValue = num('promptInfluence')
|
|
553
|
+
if (influenceValue !== undefined) setPromptInfluence(String(influenceValue))
|
|
554
|
+
const emotionValue = str('emotion')
|
|
555
|
+
if (emotionValue !== undefined) setEmotion(emotionValue)
|
|
556
|
+
const volValue = num('vol')
|
|
557
|
+
if (volValue !== undefined) setVol(String(volValue))
|
|
558
|
+
const pitchValue = num('pitch')
|
|
559
|
+
if (pitchValue !== undefined) setPitch(String(pitchValue))
|
|
560
|
+
if (Array.isArray(params.pronunciationTone)) {
|
|
561
|
+
setToneText(params.pronunciationTone.filter((item): item is string => typeof item === 'string').join('\n'))
|
|
562
|
+
}
|
|
563
|
+
const sampleRateValue = num('sampleRate')
|
|
564
|
+
if (sampleRateValue !== undefined) setSampleRate(String(sampleRateValue))
|
|
565
|
+
const bitrateValue = num('bitrate')
|
|
566
|
+
if (bitrateValue !== undefined) setBitrate(String(bitrateValue))
|
|
567
|
+
const channelValue = num('audioChannel')
|
|
568
|
+
if (channelValue !== undefined) setAudioChannel(String(channelValue))
|
|
569
|
+
const subtitleValue = bool('subtitleEnable')
|
|
570
|
+
if (subtitleValue !== undefined) setSubtitle(subtitleValue)
|
|
571
|
+
const seedValue = num('seed')
|
|
572
|
+
if (seedValue !== undefined) setSeed(String(seedValue))
|
|
573
|
+
const stepsValue = num('steps')
|
|
574
|
+
if (stepsValue !== undefined) setSteps(String(stepsValue))
|
|
575
|
+
const cfgValue = num('cfgScale')
|
|
576
|
+
if (cfgValue !== undefined) setCfgScale(String(cfgValue))
|
|
577
|
+
const previewValue = str('previewText')
|
|
578
|
+
if (previewValue !== undefined) setPreviewText(previewValue)
|
|
579
|
+
const channelIdValue = str('channelId')
|
|
580
|
+
if (channelIdValue !== undefined && modeValue === 'voice_design') setDesignChannelId(channelIdValue)
|
|
581
|
+
props.showToast('已恢复该次生成的配置,可直接再次生成')
|
|
582
|
+
}
|
|
583
|
+
|
|
584
|
+
/** 调用宿主增强(Agent 默认模型),结果先预览再应用。 */
|
|
585
|
+
const runEnhance = async (): Promise<void> => {
|
|
586
|
+
if (prompt.trim() === '') {
|
|
587
|
+
setError('请先输入文本/提示词,再点击增强')
|
|
588
|
+
return
|
|
589
|
+
}
|
|
590
|
+
setEnhancing(true)
|
|
591
|
+
setError(null)
|
|
592
|
+
try {
|
|
593
|
+
const result = await api.enhancePrompt(prompt.trim(), mode)
|
|
594
|
+
if (result.ok !== true || result.enhanced === undefined) {
|
|
595
|
+
setError(result.message ?? '增强失败,请稍后重试')
|
|
596
|
+
return
|
|
597
|
+
}
|
|
598
|
+
setEnhancePreview(result.enhanced)
|
|
599
|
+
} catch (err) {
|
|
600
|
+
setError(err instanceof Error ? err.message : String(err))
|
|
601
|
+
} finally {
|
|
602
|
+
setEnhancing(false)
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
/** 删除历史记录(对比任务卡删除该任务的全部模型条目)。 */
|
|
607
|
+
const deleteHistoryEntries = async (ids: string[]): Promise<void> => {
|
|
608
|
+
try {
|
|
609
|
+
for (const id of ids) await api.removeHistory(id)
|
|
610
|
+
} catch {
|
|
611
|
+
// best-effort
|
|
612
|
+
}
|
|
613
|
+
reload()
|
|
614
|
+
}
|
|
615
|
+
|
|
502
616
|
const openSaveDialog = (files: GeneratedAudio[], context: SaveDialogContext): void => {
|
|
503
617
|
setSaveDialog({ files, context })
|
|
504
618
|
}
|
|
@@ -767,7 +881,7 @@ export function StudioView(props: {
|
|
|
767
881
|
<div className={css.studio}>
|
|
768
882
|
<div className={css.formCol}>
|
|
769
883
|
<div className={css.modeRow}>
|
|
770
|
-
{(['tts', 'music', 'sfx', 'voice_design'] as
|
|
884
|
+
{([['tts', '🎙️'], ['music', '🎵'], ['sfx', '🔊'], ['voice_design', '🎨']] as Array<[AudioMode, string]>).map(([item, icon]) => (
|
|
771
885
|
<button
|
|
772
886
|
key={item}
|
|
773
887
|
type="button"
|
|
@@ -775,15 +889,35 @@ export function StudioView(props: {
|
|
|
775
889
|
data-active={mode === item ? 'true' : 'false'}
|
|
776
890
|
onClick={() => setMode(item)}
|
|
777
891
|
>
|
|
778
|
-
{
|
|
892
|
+
<span className={css.modeIcon}>{icon}</span>
|
|
893
|
+
{modeLabelOf(item)}
|
|
779
894
|
</button>
|
|
780
895
|
))}
|
|
781
896
|
</div>
|
|
782
897
|
|
|
898
|
+
<div className={css.formSectionRow}>
|
|
899
|
+
<p className={css.formSection}>输入</p>
|
|
900
|
+
<button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>
|
|
901
|
+
{enhancing ? '增强中…' : '✨ 增强提示词'}
|
|
902
|
+
</button>
|
|
903
|
+
</div>
|
|
783
904
|
<label className={css.label}>
|
|
784
905
|
<span>{mode === 'voice_design' ? '音色描述' : mode === 'tts' ? '文本' : '提示词'}</span>
|
|
785
906
|
<textarea className={css.textarea} value={prompt} onChange={event => setPrompt(event.target.value)} placeholder={tt('prompt.placeholder')} />
|
|
786
907
|
</label>
|
|
908
|
+
{enhancePreview !== null ? (
|
|
909
|
+
<div className={css.enhanceCard}>
|
|
910
|
+
<div className={css.enhanceCardHead}>
|
|
911
|
+
<strong>增强结果({modeLabelOf(mode)})</strong>
|
|
912
|
+
<span className={css.enhanceActions}>
|
|
913
|
+
<button type="button" className={css.ghostButton} onClick={() => { setPrompt(enhancePreview); setEnhancePreview(null); props.showToast('已应用增强结果') }}>应用</button>
|
|
914
|
+
<button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>重新生成</button>
|
|
915
|
+
<button type="button" className={css.ghostButton} onClick={() => setEnhancePreview(null)}>放弃</button>
|
|
916
|
+
</span>
|
|
917
|
+
</div>
|
|
918
|
+
<textarea className={css.textarea} value={enhancePreview} readOnly />
|
|
919
|
+
</div>
|
|
920
|
+
) : null}
|
|
787
921
|
|
|
788
922
|
{mode === 'voice_design' ? (
|
|
789
923
|
<>
|
|
@@ -806,6 +940,7 @@ export function StudioView(props: {
|
|
|
806
940
|
</>
|
|
807
941
|
) : null}
|
|
808
942
|
|
|
943
|
+
<p className={css.formSection}>模型</p>
|
|
809
944
|
{needModel ? (
|
|
810
945
|
<label className={css.checkbox} title="选择多个模型,用相同参数逐个生成,便于对比效果">
|
|
811
946
|
<input type="checkbox" checked={compareMode} onChange={event => setCompareMode(event.target.checked)} />
|
|
@@ -902,7 +1037,14 @@ export function StudioView(props: {
|
|
|
902
1037
|
)
|
|
903
1038
|
) : null}
|
|
904
1039
|
|
|
905
|
-
{globalSpecs.
|
|
1040
|
+
{globalSpecs.some(spec => spec.advanced !== true) ? <p className={css.formSection}>生成参数</p> : null}
|
|
1041
|
+
<div className={css.formFields}>
|
|
1042
|
+
{globalSpecs.filter(spec => spec.advanced !== true).map(spec => (
|
|
1043
|
+
<div key={spec.key} className={spec.key === 'lyrics' || spec.key === 'toneText' ? css.fieldFull : css.fieldCell}>
|
|
1044
|
+
{renderField(spec)}
|
|
1045
|
+
</div>
|
|
1046
|
+
))}
|
|
1047
|
+
</div>
|
|
906
1048
|
|
|
907
1049
|
{mode === 'tts' && globalSpecs.some(spec => spec.advanced === true) ? (
|
|
908
1050
|
<details className={css.advanced}>
|
|
@@ -935,6 +1077,15 @@ export function StudioView(props: {
|
|
|
935
1077
|
<span className={css.resultEmptyIcon}>🎵</span>
|
|
936
1078
|
<p>{tt('result.empty')}</p>
|
|
937
1079
|
<p className={css.resultEmptyHint}>点击「开始生成」即创建一个任务,可同时进行多个;勾选「模型对比」用多个模型同参数生成对比</p>
|
|
1080
|
+
<button type="button" className={css.ghostButton} onClick={() => {
|
|
1081
|
+
const examples: Record<AudioMode, string> = {
|
|
1082
|
+
tts: '今天是不是很开心呀(laughs),当然了!我们一起去公园散步吧。',
|
|
1083
|
+
music: 'Cinematic orchestral piece with a clear "before/after" transition at 1:00, starting minimalist piano + strings, then full orchestra entrance with timpani and brass at the 1-minute mark.',
|
|
1084
|
+
sfx: '科技感 UI 提示音:清脆短促,带轻微回声与空气感。',
|
|
1085
|
+
voice_design: '讲述悬疑故事的播音员,声音低沉富有磁性,语速时快时慢,营造紧张神秘的氛围。',
|
|
1086
|
+
}
|
|
1087
|
+
setPrompt(examples[mode] ?? '')
|
|
1088
|
+
}}>填入示例 prompt</button>
|
|
938
1089
|
</div>
|
|
939
1090
|
) : (
|
|
940
1091
|
<div className={css.taskList}>
|
|
@@ -955,6 +1106,9 @@ export function StudioView(props: {
|
|
|
955
1106
|
<span className={css.resultModeChip}>{task.mode}</span>
|
|
956
1107
|
<span className={css.taskLabel} title={task.prompt}>{task.label}</span>
|
|
957
1108
|
<span className={css.taskStatus} data-state={task.status}>{statusText}</span>
|
|
1109
|
+
{task.status === 'running' ? (
|
|
1110
|
+
<span className={css.taskBar}><i style={{ width: `${task.progress.total > 0 ? Math.round((task.progress.done / task.progress.total) * 100) : 0}%` }} /></span>
|
|
1111
|
+
) : null}
|
|
958
1112
|
<span className={css.taskActions}>
|
|
959
1113
|
{task.status === 'running' ? (
|
|
960
1114
|
<button type="button" className={css.ghostButton} onClick={() => cancelTask(task.id)}>取消</button>
|
|
@@ -1023,6 +1177,7 @@ export function StudioView(props: {
|
|
|
1023
1177
|
<summary className={css.historyCompareSummary}>
|
|
1024
1178
|
<span className={css.historyPrompt}>{item.prompt}</span>
|
|
1025
1179
|
<span className={css.historyCompareBadge}>对比 · {item.models.length} 个模型</span>
|
|
1180
|
+
<span className={css.historyTime}>{formatClock(item.createdAt)}</span>
|
|
1026
1181
|
</summary>
|
|
1027
1182
|
<div className={css.historyMeta}>{modeLabelOf(item.mode)} · {item.models.map(model => model.model).join(' / ')}</div>
|
|
1028
1183
|
{item.models.map(model => (
|
|
@@ -1040,6 +1195,15 @@ export function StudioView(props: {
|
|
|
1040
1195
|
</div>
|
|
1041
1196
|
</div>
|
|
1042
1197
|
))}
|
|
1198
|
+
<div className={css.historyActions}>
|
|
1199
|
+
<button type="button" className={css.historyIcon} title="恢复(回填配置与全部模型)" onClick={() => restoreFromParams(
|
|
1200
|
+
(item.models[0]?.entry.params ?? {}) as Record<string, unknown>,
|
|
1201
|
+
item.mode,
|
|
1202
|
+
item.models[0]?.model ?? '',
|
|
1203
|
+
item.models.map(model => model.model),
|
|
1204
|
+
)}>↺</button>
|
|
1205
|
+
<button type="button" className={css.historyIcon} title="删除整个对比任务" onClick={() => void deleteHistoryEntries(item.models.map(model => model.entry.id))}>✕</button>
|
|
1206
|
+
</div>
|
|
1043
1207
|
</details>
|
|
1044
1208
|
)
|
|
1045
1209
|
}
|
|
@@ -1048,12 +1212,19 @@ export function StudioView(props: {
|
|
|
1048
1212
|
<div className={css.historyItem} key={item.key}>
|
|
1049
1213
|
<div className={css.historyPrompt}>{entry.prompt}</div>
|
|
1050
1214
|
<div className={css.historyMeta}>{modeLabelOf(entry.mode)} · {entry.model}{entry.channel ? ` · ${entry.channel}` : ''}</div>
|
|
1215
|
+
<div className={css.historyTime}>{formatClock(entry.createdAt)}</div>
|
|
1051
1216
|
{entry.audio.map((audio, index) => (
|
|
1052
1217
|
<AudioPlayer key={index} src={audio.url} compact itemKey={`${entry.id}-${index}`} />
|
|
1053
1218
|
))}
|
|
1054
1219
|
<div className={css.historyActions}>
|
|
1055
|
-
<button type="button" className={css.
|
|
1056
|
-
<
|
|
1220
|
+
<button type="button" className={css.historyIcon} title="恢复(回填配置与 prompt)" onClick={() => restoreFromParams(
|
|
1221
|
+
(entry.params ?? {}) as Record<string, unknown>,
|
|
1222
|
+
entry.mode,
|
|
1223
|
+
entry.model,
|
|
1224
|
+
)}>↺</button>
|
|
1225
|
+
<button type="button" className={css.historyIcon} title="删除这条记录" onClick={() => void deleteHistoryEntries([entry.id])}>✕</button>
|
|
1226
|
+
<button type="button" className={css.historyIcon} title="加入资源库" onClick={() => openSaveDialog(audioRefsOfEntry(entry), contextOfEntry(entry))}>
|
|
1227
|
+
<StarIcon />
|
|
1057
1228
|
</button>
|
|
1058
1229
|
</div>
|
|
1059
1230
|
</div>
|
package/src/index.ts
CHANGED
|
@@ -16,8 +16,10 @@ import z from 'schemastery'
|
|
|
16
16
|
import type {} from '@deepseek-ai/dsh-host-webserver'
|
|
17
17
|
import type {} from '@deepseek-ai/dsh-system-prompt'
|
|
18
18
|
import type {} from '@deepseek-ai/dsh-tools'
|
|
19
|
-
import { AUDIOGEN_SETTINGS_NAMESPACE, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
19
|
+
import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
20
20
|
import { createGenerationBudget } from './audio-scheduler.ts'
|
|
21
|
+
import { enhancePromptText } from './prompt-enhance.ts'
|
|
22
|
+
import { AudioGenError } from './audio-engine.ts'
|
|
21
23
|
import { makeRoutes, type ChannelsView, type SettingsSeam } from './routes.ts'
|
|
22
24
|
import type { AudioChannel } from './audio-engine.ts'
|
|
23
25
|
import { registerAgentAudioTools, type AgentAudioToolConfig } from './agent-audio-tools.ts'
|
|
@@ -196,6 +198,13 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
196
198
|
// 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
|
|
197
199
|
const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
|
|
198
200
|
|
|
201
|
+
// 提示词增强:复用 Agent 默认模型(agent-default-model 设置),面板与 Agent 工具共用。
|
|
202
|
+
const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
|
|
203
|
+
const seam = ctx.get('settings') as unknown as SettingsSeam
|
|
204
|
+
if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
|
|
205
|
+
return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
|
|
206
|
+
}
|
|
207
|
+
|
|
199
208
|
const channelsView = (): ChannelsView => {
|
|
200
209
|
const value = resolve()
|
|
201
210
|
return { channels: value.channels, defaultChannelId: value.defaultChannelId }
|
|
@@ -209,6 +218,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
209
218
|
resolveChannels: channelsView,
|
|
210
219
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
211
220
|
budget,
|
|
221
|
+
enhance,
|
|
212
222
|
})
|
|
213
223
|
const disposers = routes.map(route => ctx.webServer.register(route))
|
|
214
224
|
return () => { for (const dispose of disposers) dispose() }
|
|
@@ -225,6 +235,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
225
235
|
defaultChannelId: value.defaultChannelId,
|
|
226
236
|
autoSaveToLibrary: value.autoSaveToLibrary,
|
|
227
237
|
budget,
|
|
238
|
+
enhance,
|
|
228
239
|
}
|
|
229
240
|
}), 'dsh-audiogen: agent audio tools')
|
|
230
241
|
})
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 提示词增强:复用 Agent 当前默认模型(设置「模型」里的 provider/model,
|
|
3
|
+
* 即 agent-default-model 命名空间),宿主端发起一次 LLM 调用把用户 prompt
|
|
4
|
+
* 扩展成更适合生成的任务描述。不新增 API key 配置。
|
|
5
|
+
*
|
|
6
|
+
* 面板「✨ 增强提示词」与 generate_audio 工具的 enhance_prompt 都走这里。
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { AudioMode } from './protocol.ts'
|
|
10
|
+
import { AudioGenError } from './audio-engine.ts'
|
|
11
|
+
|
|
12
|
+
/** 按生成模式给出增强指令(系统提示)。 */
|
|
13
|
+
function instructionsFor(mode: AudioMode): string {
|
|
14
|
+
const common = [
|
|
15
|
+
'你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,',
|
|
16
|
+
'请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。',
|
|
17
|
+
'只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。',
|
|
18
|
+
'保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。',
|
|
19
|
+
'描述控制在 200-600 字左右。',
|
|
20
|
+
].join('')
|
|
21
|
+
const perMode: Record<AudioMode, string> = {
|
|
22
|
+
tts: '这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。',
|
|
23
|
+
music: '这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。',
|
|
24
|
+
sfx: '这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。',
|
|
25
|
+
voice_design: '这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。',
|
|
26
|
+
}
|
|
27
|
+
return common + perMode[mode]
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface PromptEnhanceDeps {
|
|
31
|
+
/** DSH 设置 seam(读 agent-default-model)。 */
|
|
32
|
+
settings: { describe(options?: { redactSecrets?: boolean }): Array<{ ns: unknown; value?: unknown }> }
|
|
33
|
+
/** 宿主 LLM 运行时访问器(延迟读取,调用时才获取)。 */
|
|
34
|
+
llm?: () => unknown
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
|
|
38
|
+
export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string, mode: AudioMode): Promise<string> {
|
|
39
|
+
const text = prompt.trim()
|
|
40
|
+
if (text === '') throw new AudioGenError('提示词为空,无法增强', 'enhance-empty-prompt')
|
|
41
|
+
const descriptor = (deps.settings.describe({ redactSecrets: true }) ?? []).find(candidate => String(candidate.ns) === 'agent-default-model')
|
|
42
|
+
const value = (descriptor?.value ?? {}) as { provider?: unknown; model?: unknown }
|
|
43
|
+
const provider = typeof value.provider === 'string' && value.provider.trim() !== '' ? value.provider.trim() : ''
|
|
44
|
+
const model = typeof value.model === 'string' && value.model.trim() !== '' ? value.model.trim() : ''
|
|
45
|
+
if (provider === '' || model === '') {
|
|
46
|
+
throw new AudioGenError('未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型', 'no-default-model')
|
|
47
|
+
}
|
|
48
|
+
const runtime = deps.llm?.() as { stream?: (options: unknown) => AsyncIterable<unknown> } | undefined
|
|
49
|
+
if (runtime === undefined || runtime.stream === undefined) {
|
|
50
|
+
throw new AudioGenError('宿主 LLM 服务不可用(ctx.llm 未注册)', 'llm-unavailable')
|
|
51
|
+
}
|
|
52
|
+
let output = ''
|
|
53
|
+
for await (const chunk of runtime.stream({
|
|
54
|
+
provider,
|
|
55
|
+
model,
|
|
56
|
+
messages: [{ role: 'user', content: text }],
|
|
57
|
+
system: instructionsFor(mode),
|
|
58
|
+
temperature: 0.7,
|
|
59
|
+
maxTokens: 1200,
|
|
60
|
+
})) {
|
|
61
|
+
const record = chunk as { type?: string; text?: string }
|
|
62
|
+
if (record.type === 'text-delta' && typeof record.text === 'string') output += record.text
|
|
63
|
+
}
|
|
64
|
+
return stripFences(output.trim())
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** 去掉模型可能包裹的 ``` 代码围栏。 */
|
|
68
|
+
function stripFences(value: string): string {
|
|
69
|
+
if (value === '') return value
|
|
70
|
+
const withoutFence = value.replace(/^```[a-zA-Z]*\s*\n?/, '').replace(/\n?```\s*$/, '')
|
|
71
|
+
return withoutFence.trim()
|
|
72
|
+
}
|
package/src/protocol.ts
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
|
|
9
9
|
|
|
10
10
|
/** Published package version shared by the host updater and the client UI. */
|
|
11
|
-
export const PLUGIN_VERSION = '0.4.
|
|
11
|
+
export const PLUGIN_VERSION = '0.4.5'
|
|
12
12
|
|
|
13
13
|
/** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
|
|
14
14
|
export const SETTINGS_API = {
|
|
@@ -24,6 +24,9 @@ export const TASK_API = {
|
|
|
24
24
|
cancel: '/api/dsh-audiogen/task/cancel',
|
|
25
25
|
} as const
|
|
26
26
|
|
|
27
|
+
/** Loopback-only prompt enhancement route (uses the agent's default model). */
|
|
28
|
+
export const ENHANCE_API = '/api/dsh-audiogen/prompt/enhance' as const
|
|
29
|
+
|
|
27
30
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
28
31
|
export const PRESETS_API = '/api/dsh-audiogen/presets' as const
|
|
29
32
|
|
package/src/routes.ts
CHANGED
|
@@ -16,7 +16,7 @@ import { discoverAudioModels } from './audio-models.ts'
|
|
|
16
16
|
import { AUDIO_PRESETS } from './audio-presets.ts'
|
|
17
17
|
import { appendHistory, clearHistory, listHistory, readAudioFile, removeHistory, saveAudioFile, listLibrary, saveToLibrary, updateLibraryEntry, removeLibraryEntries, readLibraryFile } from './audio-store.ts'
|
|
18
18
|
import {
|
|
19
|
-
AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
|
|
19
|
+
AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
|
|
20
20
|
LIBRARY_TYPES,
|
|
21
21
|
type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput, type LibraryAudioInput, type LibraryProvenance, type LibraryType,
|
|
22
22
|
} from './protocol.ts'
|
|
@@ -47,6 +47,8 @@ export interface AudiogenRoutesDeps {
|
|
|
47
47
|
autoSave: () => boolean
|
|
48
48
|
/** Global upstream concurrency gate (maxConcurrentGenerations). */
|
|
49
49
|
budget: GenerationBudget
|
|
50
|
+
/** 提示词增强:调用 Agent 默认模型,返回增强后的文本。 */
|
|
51
|
+
enhance: (prompt: string, mode: GenerateAudioRequest['mode']) => Promise<string>
|
|
50
52
|
}
|
|
51
53
|
|
|
52
54
|
function isLoopbackRequest(request: IncomingMessage): boolean {
|
|
@@ -528,6 +530,27 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
|
|
|
528
530
|
writeJson(res, 200, { ok: true, aborted: controllers !== undefined ? controllers.size : 0 })
|
|
529
531
|
},
|
|
530
532
|
},
|
|
533
|
+
// ------------------------------------------------------- prompt enhance
|
|
534
|
+
{
|
|
535
|
+
kind: 'exact',
|
|
536
|
+
path: ENHANCE_API,
|
|
537
|
+
handler: async (req, res) => {
|
|
538
|
+
if (!guard(req, res, 'POST')) return
|
|
539
|
+
const body = await readJsonBody(req)
|
|
540
|
+
const prompt = typeof body?.prompt === 'string' ? body.prompt.trim() : ''
|
|
541
|
+
if (prompt === '') {
|
|
542
|
+
writeJson(res, 200, { ok: false, code: 'bad-request', message: 'prompt is required' })
|
|
543
|
+
return
|
|
544
|
+
}
|
|
545
|
+
const mode = body?.mode === 'music' ? 'music' : body?.mode === 'sfx' ? 'sfx' : body?.mode === 'voice_design' ? 'voice_design' : 'tts'
|
|
546
|
+
try {
|
|
547
|
+
const enhanced = await deps.enhance(prompt, mode)
|
|
548
|
+
writeJson(res, 200, { ok: true, enhanced })
|
|
549
|
+
} catch (error) {
|
|
550
|
+
writeJson(res, 200, { ok: false, code: 'enhance-failed', message: messageOf(error) })
|
|
551
|
+
}
|
|
552
|
+
},
|
|
553
|
+
},
|
|
531
554
|
// ----------------------------------------------------------- audio file
|
|
532
555
|
{
|
|
533
556
|
kind: 'prefix',
|