dsh-audiogen 0.4.4 → 0.4.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +432 -338
- package/lib/client.js.map +1 -1
- package/lib/index.js +100 -2
- package/package.json +1 -1
- package/src/agent-audio-tools.ts +11 -0
- package/src/client/api.ts +8 -1
- package/src/client/audio-panel.module.css +94 -11
- package/src/client/studio-view.tsx +44 -1
- package/src/index.ts +12 -1
- package/src/prompt-enhance.ts +72 -0
- package/src/protocol.ts +4 -1
- package/src/routes.ts +24 -1
package/lib/index.js
CHANGED
|
@@ -24,6 +24,8 @@ const SETTINGS_API = {
|
|
|
24
24
|
const GENERATE_API = "/api/dsh-audiogen/generate";
|
|
25
25
|
/** Loopback-only task cancellation route (aborts the host-side upstream call). */
|
|
26
26
|
const TASK_API = { cancel: "/api/dsh-audiogen/task/cancel" };
|
|
27
|
+
/** Loopback-only prompt enhancement route (uses the agent's default model). */
|
|
28
|
+
const ENHANCE_API = "/api/dsh-audiogen/prompt/enhance";
|
|
27
29
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
28
30
|
const PRESETS_API = "/api/dsh-audiogen/presets";
|
|
29
31
|
/** Host-mediated model/voice discovery endpoint. */
|
|
@@ -864,6 +866,55 @@ async function generateAudio(channel, request, signal) {
|
|
|
864
866
|
return genericAudio(channel, request, signal);
|
|
865
867
|
}
|
|
866
868
|
//#endregion
|
|
869
|
+
//#region src/prompt-enhance.ts
|
|
870
|
+
/** 按生成模式给出增强指令(系统提示)。 */
|
|
871
|
+
function instructionsFor(mode) {
|
|
872
|
+
return [
|
|
873
|
+
"你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,",
|
|
874
|
+
"请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。",
|
|
875
|
+
"只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。",
|
|
876
|
+
"保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。",
|
|
877
|
+
"描述控制在 200-600 字左右。"
|
|
878
|
+
].join("") + {
|
|
879
|
+
tts: "这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。",
|
|
880
|
+
music: "这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。",
|
|
881
|
+
sfx: "这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。",
|
|
882
|
+
voice_design: "这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。"
|
|
883
|
+
}[mode];
|
|
884
|
+
}
|
|
885
|
+
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
|
|
886
|
+
async function enhancePromptText(deps, prompt, mode) {
|
|
887
|
+
const text = prompt.trim();
|
|
888
|
+
if (text === "") throw new AudioGenError("提示词为空,无法增强", "enhance-empty-prompt");
|
|
889
|
+
const value = (deps.settings.describe({ redactSecrets: true }) ?? []).find((candidate) => String(candidate.ns) === "agent-default-model")?.value ?? {};
|
|
890
|
+
const provider = typeof value.provider === "string" && value.provider.trim() !== "" ? value.provider.trim() : "";
|
|
891
|
+
const model = typeof value.model === "string" && value.model.trim() !== "" ? value.model.trim() : "";
|
|
892
|
+
if (provider === "" || model === "") throw new AudioGenError("未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型", "no-default-model");
|
|
893
|
+
const runtime = deps.llm?.();
|
|
894
|
+
if (runtime === void 0 || runtime.stream === void 0) throw new AudioGenError("宿主 LLM 服务不可用(ctx.llm 未注册)", "llm-unavailable");
|
|
895
|
+
let output = "";
|
|
896
|
+
for await (const chunk of runtime.stream({
|
|
897
|
+
provider,
|
|
898
|
+
model,
|
|
899
|
+
messages: [{
|
|
900
|
+
role: "user",
|
|
901
|
+
content: text
|
|
902
|
+
}],
|
|
903
|
+
system: instructionsFor(mode),
|
|
904
|
+
temperature: .7,
|
|
905
|
+
maxTokens: 1200
|
|
906
|
+
})) {
|
|
907
|
+
const record = chunk;
|
|
908
|
+
if (record.type === "text-delta" && typeof record.text === "string") output += record.text;
|
|
909
|
+
}
|
|
910
|
+
return stripFences(output.trim());
|
|
911
|
+
}
|
|
912
|
+
/** 去掉模型可能包裹的 ``` 代码围栏。 */
|
|
913
|
+
function stripFences(value) {
|
|
914
|
+
if (value === "") return value;
|
|
915
|
+
return value.replace(/^```[a-zA-Z]*\s*\n?/, "").replace(/\n?```\s*$/, "").trim();
|
|
916
|
+
}
|
|
917
|
+
//#endregion
|
|
867
918
|
//#region src/audio-presets.ts
|
|
868
919
|
const AUDIO_PRESETS = [
|
|
869
920
|
{
|
|
@@ -2046,6 +2097,36 @@ function makeRoutes(deps) {
|
|
|
2046
2097
|
});
|
|
2047
2098
|
}
|
|
2048
2099
|
},
|
|
2100
|
+
{
|
|
2101
|
+
kind: "exact",
|
|
2102
|
+
path: ENHANCE_API,
|
|
2103
|
+
handler: async (req, res) => {
|
|
2104
|
+
if (!guard(req, res, "POST")) return;
|
|
2105
|
+
const body = await readJsonBody(req);
|
|
2106
|
+
const prompt = typeof body?.prompt === "string" ? body.prompt.trim() : "";
|
|
2107
|
+
if (prompt === "") {
|
|
2108
|
+
writeJson(res, 200, {
|
|
2109
|
+
ok: false,
|
|
2110
|
+
code: "bad-request",
|
|
2111
|
+
message: "prompt is required"
|
|
2112
|
+
});
|
|
2113
|
+
return;
|
|
2114
|
+
}
|
|
2115
|
+
const mode = body?.mode === "music" ? "music" : body?.mode === "sfx" ? "sfx" : body?.mode === "voice_design" ? "voice_design" : "tts";
|
|
2116
|
+
try {
|
|
2117
|
+
writeJson(res, 200, {
|
|
2118
|
+
ok: true,
|
|
2119
|
+
enhanced: await deps.enhance(prompt, mode)
|
|
2120
|
+
});
|
|
2121
|
+
} catch (error) {
|
|
2122
|
+
writeJson(res, 200, {
|
|
2123
|
+
ok: false,
|
|
2124
|
+
code: "enhance-failed",
|
|
2125
|
+
message: messageOf(error)
|
|
2126
|
+
});
|
|
2127
|
+
}
|
|
2128
|
+
}
|
|
2129
|
+
},
|
|
2049
2130
|
{
|
|
2050
2131
|
kind: "prefix",
|
|
2051
2132
|
path: AUDIO_API.file,
|
|
@@ -2483,6 +2564,10 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
2483
2564
|
type: "string",
|
|
2484
2565
|
description: "Optional preview text for voice_design."
|
|
2485
2566
|
},
|
|
2567
|
+
enhance_prompt: {
|
|
2568
|
+
type: "boolean",
|
|
2569
|
+
description: "Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used."
|
|
2570
|
+
},
|
|
2486
2571
|
speed: {
|
|
2487
2572
|
type: "number",
|
|
2488
2573
|
description: "Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1)."
|
|
@@ -2725,6 +2810,9 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
2725
2810
|
/** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
|
|
2726
2811
|
const runOne = async (picked) => {
|
|
2727
2812
|
const request = buildRequest(picked);
|
|
2813
|
+
if (args.enhance_prompt === true && config.enhance !== void 0) try {
|
|
2814
|
+
request.prompt = await config.enhance(request.prompt, request.mode);
|
|
2815
|
+
} catch {}
|
|
2728
2816
|
try {
|
|
2729
2817
|
const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => {}));
|
|
2730
2818
|
let outputs;
|
|
@@ -3137,6 +3225,14 @@ function apply(ctx, config) {
|
|
|
3137
3225
|
};
|
|
3138
3226
|
};
|
|
3139
3227
|
const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations);
|
|
3228
|
+
const enhance = async (prompt, mode) => {
|
|
3229
|
+
const seam = ctx.get("settings");
|
|
3230
|
+
if (seam?.describe === void 0) throw new AudioGenError("设置服务不可用,无法增强提示词", "settings-unavailable");
|
|
3231
|
+
return enhancePromptText({
|
|
3232
|
+
settings: seam,
|
|
3233
|
+
llm: () => ctx.get("llm")
|
|
3234
|
+
}, prompt, mode);
|
|
3235
|
+
};
|
|
3140
3236
|
const channelsView = () => {
|
|
3141
3237
|
const value = resolve();
|
|
3142
3238
|
return {
|
|
@@ -3151,7 +3247,8 @@ function apply(ctx, config) {
|
|
|
3151
3247
|
settings: seam,
|
|
3152
3248
|
resolveChannels: channelsView,
|
|
3153
3249
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
3154
|
-
budget
|
|
3250
|
+
budget,
|
|
3251
|
+
enhance
|
|
3155
3252
|
}).map((route) => ctx.webServer.register(route));
|
|
3156
3253
|
return () => {
|
|
3157
3254
|
for (const dispose of disposers) dispose();
|
|
@@ -3167,7 +3264,8 @@ function apply(ctx, config) {
|
|
|
3167
3264
|
channels: value.channels,
|
|
3168
3265
|
defaultChannelId: value.defaultChannelId,
|
|
3169
3266
|
autoSaveToLibrary: value.autoSaveToLibrary,
|
|
3170
|
-
budget
|
|
3267
|
+
budget,
|
|
3268
|
+
enhance
|
|
3171
3269
|
};
|
|
3172
3270
|
}), "dsh-audiogen: agent audio tools");
|
|
3173
3271
|
});
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-audiogen",
|
|
3
3
|
"description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.5",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|
package/src/agent-audio-tools.ts
CHANGED
|
@@ -21,6 +21,8 @@ export interface AgentAudioToolConfig {
|
|
|
21
21
|
autoSaveToLibrary: boolean
|
|
22
22
|
/** 全局并发闸门(与面板路由共享「最大并发生成数」)。 */
|
|
23
23
|
budget?: GenerationBudget
|
|
24
|
+
/** 提示词增强(复用 Agent 默认模型);enhance_prompt=true 时在生成前调用。 */
|
|
25
|
+
enhance?: (prompt: string, mode: AudioMode) => Promise<string>
|
|
24
26
|
}
|
|
25
27
|
|
|
26
28
|
interface AgentAudioRef {
|
|
@@ -157,6 +159,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
157
159
|
},
|
|
158
160
|
voice: { type: 'string', description: 'Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio.' },
|
|
159
161
|
preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
|
|
162
|
+
enhance_prompt: { type: 'boolean', description: 'Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used.' },
|
|
160
163
|
speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
|
|
161
164
|
duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
|
|
162
165
|
lyrics: { type: 'string', description: 'Lyrics for music generation (MiniMax music-3.0/music-cover). Required unless is_instrumental is true. Split verses with an empty line.' },
|
|
@@ -302,6 +305,14 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
302
305
|
/** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
|
|
303
306
|
const runOne = async (picked: { channel: AudioChannel; alias: string; upstream: string }): Promise<AgentAudioGroup> => {
|
|
304
307
|
const request = buildRequest(picked)
|
|
308
|
+
// 可选:生成前用 Agent 默认模型增强 prompt(失败则沿用原文)
|
|
309
|
+
if (args.enhance_prompt === true && config.enhance !== undefined) {
|
|
310
|
+
try {
|
|
311
|
+
request.prompt = await config.enhance(request.prompt, request.mode)
|
|
312
|
+
} catch {
|
|
313
|
+
// 增强失败不阻断生成
|
|
314
|
+
}
|
|
315
|
+
}
|
|
305
316
|
try {
|
|
306
317
|
// 与面板路由共享全局并发闸门(限流时排队;取消时立即出队)。
|
|
307
318
|
const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => { /* 默认不限制 */ }))
|
package/src/client/api.ts
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
6
|
import {
|
|
7
|
-
GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
|
|
7
|
+
ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
|
|
8
8
|
type GenerateAudioRequest, type GeneratedAudio, type HistoryEntry,
|
|
9
9
|
type LibraryEntry, type LibrarySaveRequest, type LibraryUpdateRequest,
|
|
10
10
|
} from '../protocol.ts'
|
|
@@ -42,6 +42,13 @@ export class AudiogenApi {
|
|
|
42
42
|
await postJson(TASK_API.cancel, { taskId }).catch(() => { /* best-effort */ })
|
|
43
43
|
}
|
|
44
44
|
|
|
45
|
+
/** 提示词增强(复用 Agent 默认模型)。 */
|
|
46
|
+
async enhancePrompt(prompt: string, mode: string): Promise<{ ok: boolean; enhanced?: string; code?: string; message?: string }> {
|
|
47
|
+
const response = await postJson(ENHANCE_API, { prompt, mode })
|
|
48
|
+
const body = await response.json() as { ok?: boolean; enhanced?: string; code?: string; message?: string }
|
|
49
|
+
return { ok: body.ok === true, ...(body.enhanced === undefined ? {} : { enhanced: body.enhanced }), ...(body.code === undefined ? {} : { code: body.code }), ...(body.message === undefined ? {} : { message: body.message }) }
|
|
50
|
+
}
|
|
51
|
+
|
|
45
52
|
async history(): Promise<HistoryEntry[]> {
|
|
46
53
|
const response = await postJson(HISTORY_API.list, {})
|
|
47
54
|
const body = await response.json() as { ok?: boolean; history?: HistoryEntry[] }
|
|
@@ -111,8 +111,8 @@
|
|
|
111
111
|
.formCol::-webkit-scrollbar-thumb { background: var(--dsw-alias-border-l2); border-radius: 999px; }
|
|
112
112
|
|
|
113
113
|
.modeRow {
|
|
114
|
-
display:
|
|
115
|
-
|
|
114
|
+
display: flex;
|
|
115
|
+
flex-wrap: wrap;
|
|
116
116
|
gap: 6px;
|
|
117
117
|
}
|
|
118
118
|
|
|
@@ -120,8 +120,10 @@
|
|
|
120
120
|
display: flex;
|
|
121
121
|
align-items: center;
|
|
122
122
|
justify-content: center;
|
|
123
|
+
flex: 1 1 auto;
|
|
124
|
+
min-width: 0;
|
|
123
125
|
min-height: 34px;
|
|
124
|
-
padding: 6px
|
|
126
|
+
padding: 6px 10px;
|
|
125
127
|
border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
|
|
126
128
|
border-radius: 9px;
|
|
127
129
|
background: var(--dsw-alias-bg-layer-1, #fff);
|
|
@@ -960,7 +962,7 @@
|
|
|
960
962
|
|
|
961
963
|
.resultGroupError {
|
|
962
964
|
font-size: 12px;
|
|
963
|
-
color: #dc2626;
|
|
965
|
+
color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
964
966
|
}
|
|
965
967
|
|
|
966
968
|
.resultGroupCount {
|
|
@@ -1021,7 +1023,7 @@
|
|
|
1021
1023
|
}
|
|
1022
1024
|
|
|
1023
1025
|
.taskCard[data-state='failed'] {
|
|
1024
|
-
border-color: #dc2626;
|
|
1026
|
+
border-color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
1025
1027
|
}
|
|
1026
1028
|
|
|
1027
1029
|
.taskCard[data-state='cancelled'] {
|
|
@@ -1051,11 +1053,11 @@
|
|
|
1051
1053
|
}
|
|
1052
1054
|
|
|
1053
1055
|
.taskStatus[data-state='done'] {
|
|
1054
|
-
color: #16a34a;
|
|
1056
|
+
color: var(--dsw-alias-state-success-primary, #16a34a);
|
|
1055
1057
|
}
|
|
1056
1058
|
|
|
1057
1059
|
.taskStatus[data-state='failed'] {
|
|
1058
|
-
color: #dc2626;
|
|
1060
|
+
color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
1059
1061
|
}
|
|
1060
1062
|
|
|
1061
1063
|
.taskActions {
|
|
@@ -1107,11 +1109,11 @@
|
|
|
1107
1109
|
.historyCompareBadge {
|
|
1108
1110
|
font-size: 11px;
|
|
1109
1111
|
font-weight: 600;
|
|
1110
|
-
color: #
|
|
1111
|
-
border: 1px solid #
|
|
1112
|
+
color: var(--dsw-alias-label-secondary, #6b7280);
|
|
1113
|
+
border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
|
|
1112
1114
|
border-radius: 999px;
|
|
1113
1115
|
padding: 1px 8px;
|
|
1114
|
-
background: #
|
|
1116
|
+
background: var(--dsw-alias-bg-layer-3, #fff);
|
|
1115
1117
|
white-space: nowrap;
|
|
1116
1118
|
}
|
|
1117
1119
|
|
|
@@ -1221,7 +1223,7 @@
|
|
|
1221
1223
|
}
|
|
1222
1224
|
|
|
1223
1225
|
.historyIcon[title^='删除']:hover {
|
|
1224
|
-
color: #dc2626;
|
|
1226
|
+
color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
1225
1227
|
}
|
|
1226
1228
|
|
|
1227
1229
|
/* 历史 prompt 两行截断 */
|
|
@@ -1231,3 +1233,84 @@
|
|
|
1231
1233
|
-webkit-box-orient: vertical;
|
|
1232
1234
|
overflow: hidden;
|
|
1233
1235
|
}
|
|
1236
|
+
|
|
1237
|
+
/* ------------------------------------------------------- P3 打磨:响应式与统一质感 */
|
|
1238
|
+
|
|
1239
|
+
/* 焦点态统一(键盘可达性) */
|
|
1240
|
+
.input:focus-visible,
|
|
1241
|
+
.select:focus-visible,
|
|
1242
|
+
.textarea:focus-visible,
|
|
1243
|
+
.modeButton:focus-visible,
|
|
1244
|
+
.ghostButton:focus-visible,
|
|
1245
|
+
.historyAction:focus-visible,
|
|
1246
|
+
.historyIcon:focus-visible {
|
|
1247
|
+
outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
|
|
1248
|
+
outline-offset: 1px;
|
|
1249
|
+
}
|
|
1250
|
+
|
|
1251
|
+
.generate:focus-visible {
|
|
1252
|
+
outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
|
|
1253
|
+
outline-offset: 2px;
|
|
1254
|
+
}
|
|
1255
|
+
|
|
1256
|
+
/* 中窄屏:历史移到下方整行 */
|
|
1257
|
+
@media (max-width: 1100px) {
|
|
1258
|
+
.studio {
|
|
1259
|
+
flex-wrap: wrap;
|
|
1260
|
+
}
|
|
1261
|
+
|
|
1262
|
+
.historyCol {
|
|
1263
|
+
width: 100%;
|
|
1264
|
+
min-width: 0;
|
|
1265
|
+
max-width: none;
|
|
1266
|
+
}
|
|
1267
|
+
}
|
|
1268
|
+
|
|
1269
|
+
/* 窄屏:单列堆叠 */
|
|
1270
|
+
@media (max-width: 720px) {
|
|
1271
|
+
.formCol {
|
|
1272
|
+
width: 100%;
|
|
1273
|
+
min-width: 0;
|
|
1274
|
+
max-width: none;
|
|
1275
|
+
}
|
|
1276
|
+
|
|
1277
|
+
.resultCol {
|
|
1278
|
+
min-height: 320px;
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
|
|
1282
|
+
/* 提示词增强 */
|
|
1283
|
+
.formSectionRow {
|
|
1284
|
+
display: flex;
|
|
1285
|
+
align-items: flex-end;
|
|
1286
|
+
justify-content: space-between;
|
|
1287
|
+
gap: 8px;
|
|
1288
|
+
}
|
|
1289
|
+
|
|
1290
|
+
.formSectionRow .formSection {
|
|
1291
|
+
flex: 1;
|
|
1292
|
+
}
|
|
1293
|
+
|
|
1294
|
+
.enhanceCard {
|
|
1295
|
+
border: 1px solid var(--dsw-alias-border-l1, #e5e7eb);
|
|
1296
|
+
border-radius: 10px;
|
|
1297
|
+
padding: 10px;
|
|
1298
|
+
background: var(--dsw-alias-bg-layer-2, #fafafa);
|
|
1299
|
+
display: flex;
|
|
1300
|
+
flex-direction: column;
|
|
1301
|
+
gap: 8px;
|
|
1302
|
+
}
|
|
1303
|
+
|
|
1304
|
+
.enhanceCardHead {
|
|
1305
|
+
display: flex;
|
|
1306
|
+
align-items: center;
|
|
1307
|
+
justify-content: space-between;
|
|
1308
|
+
gap: 8px;
|
|
1309
|
+
font-size: 12px;
|
|
1310
|
+
color: var(--dsw-alias-label-secondary, #6b7280);
|
|
1311
|
+
}
|
|
1312
|
+
|
|
1313
|
+
.enhanceActions {
|
|
1314
|
+
display: flex;
|
|
1315
|
+
gap: 6px;
|
|
1316
|
+
}
|
|
@@ -230,6 +230,9 @@ export function StudioView(props: {
|
|
|
230
230
|
const [bitrate, setBitrate] = useState('')
|
|
231
231
|
const [audioChannel, setAudioChannel] = useState('')
|
|
232
232
|
const [subtitle, setSubtitle] = useState(false)
|
|
233
|
+
// 提示词增强
|
|
234
|
+
const [enhancing, setEnhancing] = useState(false)
|
|
235
|
+
const [enhancePreview, setEnhancePreview] = useState<string | null>(null)
|
|
233
236
|
// Stable Audio 参数(仅 Stability 渠道显示)
|
|
234
237
|
const [seed, setSeed] = useState('')
|
|
235
238
|
const [steps, setSteps] = useState('')
|
|
@@ -578,6 +581,28 @@ export function StudioView(props: {
|
|
|
578
581
|
props.showToast('已恢复该次生成的配置,可直接再次生成')
|
|
579
582
|
}
|
|
580
583
|
|
|
584
|
+
/** 调用宿主增强(Agent 默认模型),结果先预览再应用。 */
|
|
585
|
+
const runEnhance = async (): Promise<void> => {
|
|
586
|
+
if (prompt.trim() === '') {
|
|
587
|
+
setError('请先输入文本/提示词,再点击增强')
|
|
588
|
+
return
|
|
589
|
+
}
|
|
590
|
+
setEnhancing(true)
|
|
591
|
+
setError(null)
|
|
592
|
+
try {
|
|
593
|
+
const result = await api.enhancePrompt(prompt.trim(), mode)
|
|
594
|
+
if (result.ok !== true || result.enhanced === undefined) {
|
|
595
|
+
setError(result.message ?? '增强失败,请稍后重试')
|
|
596
|
+
return
|
|
597
|
+
}
|
|
598
|
+
setEnhancePreview(result.enhanced)
|
|
599
|
+
} catch (err) {
|
|
600
|
+
setError(err instanceof Error ? err.message : String(err))
|
|
601
|
+
} finally {
|
|
602
|
+
setEnhancing(false)
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
|
|
581
606
|
/** 删除历史记录(对比任务卡删除该任务的全部模型条目)。 */
|
|
582
607
|
const deleteHistoryEntries = async (ids: string[]): Promise<void> => {
|
|
583
608
|
try {
|
|
@@ -870,11 +895,29 @@ export function StudioView(props: {
|
|
|
870
895
|
))}
|
|
871
896
|
</div>
|
|
872
897
|
|
|
873
|
-
<
|
|
898
|
+
<div className={css.formSectionRow}>
|
|
899
|
+
<p className={css.formSection}>输入</p>
|
|
900
|
+
<button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>
|
|
901
|
+
{enhancing ? '增强中…' : '✨ 增强提示词'}
|
|
902
|
+
</button>
|
|
903
|
+
</div>
|
|
874
904
|
<label className={css.label}>
|
|
875
905
|
<span>{mode === 'voice_design' ? '音色描述' : mode === 'tts' ? '文本' : '提示词'}</span>
|
|
876
906
|
<textarea className={css.textarea} value={prompt} onChange={event => setPrompt(event.target.value)} placeholder={tt('prompt.placeholder')} />
|
|
877
907
|
</label>
|
|
908
|
+
{enhancePreview !== null ? (
|
|
909
|
+
<div className={css.enhanceCard}>
|
|
910
|
+
<div className={css.enhanceCardHead}>
|
|
911
|
+
<strong>增强结果({modeLabelOf(mode)})</strong>
|
|
912
|
+
<span className={css.enhanceActions}>
|
|
913
|
+
<button type="button" className={css.ghostButton} onClick={() => { setPrompt(enhancePreview); setEnhancePreview(null); props.showToast('已应用增强结果') }}>应用</button>
|
|
914
|
+
<button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>重新生成</button>
|
|
915
|
+
<button type="button" className={css.ghostButton} onClick={() => setEnhancePreview(null)}>放弃</button>
|
|
916
|
+
</span>
|
|
917
|
+
</div>
|
|
918
|
+
<textarea className={css.textarea} value={enhancePreview} readOnly />
|
|
919
|
+
</div>
|
|
920
|
+
) : null}
|
|
878
921
|
|
|
879
922
|
{mode === 'voice_design' ? (
|
|
880
923
|
<>
|
package/src/index.ts
CHANGED
|
@@ -16,8 +16,10 @@ import z from 'schemastery'
|
|
|
16
16
|
import type {} from '@deepseek-ai/dsh-host-webserver'
|
|
17
17
|
import type {} from '@deepseek-ai/dsh-system-prompt'
|
|
18
18
|
import type {} from '@deepseek-ai/dsh-tools'
|
|
19
|
-
import { AUDIOGEN_SETTINGS_NAMESPACE, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
19
|
+
import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
20
20
|
import { createGenerationBudget } from './audio-scheduler.ts'
|
|
21
|
+
import { enhancePromptText } from './prompt-enhance.ts'
|
|
22
|
+
import { AudioGenError } from './audio-engine.ts'
|
|
21
23
|
import { makeRoutes, type ChannelsView, type SettingsSeam } from './routes.ts'
|
|
22
24
|
import type { AudioChannel } from './audio-engine.ts'
|
|
23
25
|
import { registerAgentAudioTools, type AgentAudioToolConfig } from './agent-audio-tools.ts'
|
|
@@ -196,6 +198,13 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
196
198
|
// 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
|
|
197
199
|
const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
|
|
198
200
|
|
|
201
|
+
// 提示词增强:复用 Agent 默认模型(agent-default-model 设置),面板与 Agent 工具共用。
|
|
202
|
+
const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
|
|
203
|
+
const seam = ctx.get('settings') as unknown as SettingsSeam
|
|
204
|
+
if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
|
|
205
|
+
return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
|
|
206
|
+
}
|
|
207
|
+
|
|
199
208
|
const channelsView = (): ChannelsView => {
|
|
200
209
|
const value = resolve()
|
|
201
210
|
return { channels: value.channels, defaultChannelId: value.defaultChannelId }
|
|
@@ -209,6 +218,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
209
218
|
resolveChannels: channelsView,
|
|
210
219
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
211
220
|
budget,
|
|
221
|
+
enhance,
|
|
212
222
|
})
|
|
213
223
|
const disposers = routes.map(route => ctx.webServer.register(route))
|
|
214
224
|
return () => { for (const dispose of disposers) dispose() }
|
|
@@ -225,6 +235,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
225
235
|
defaultChannelId: value.defaultChannelId,
|
|
226
236
|
autoSaveToLibrary: value.autoSaveToLibrary,
|
|
227
237
|
budget,
|
|
238
|
+
enhance,
|
|
228
239
|
}
|
|
229
240
|
}), 'dsh-audiogen: agent audio tools')
|
|
230
241
|
})
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 提示词增强:复用 Agent 当前默认模型(设置「模型」里的 provider/model,
|
|
3
|
+
* 即 agent-default-model 命名空间),宿主端发起一次 LLM 调用把用户 prompt
|
|
4
|
+
* 扩展成更适合生成的任务描述。不新增 API key 配置。
|
|
5
|
+
*
|
|
6
|
+
* 面板「✨ 增强提示词」与 generate_audio 工具的 enhance_prompt 都走这里。
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { AudioMode } from './protocol.ts'
|
|
10
|
+
import { AudioGenError } from './audio-engine.ts'
|
|
11
|
+
|
|
12
|
+
/** 按生成模式给出增强指令(系统提示)。 */
|
|
13
|
+
function instructionsFor(mode: AudioMode): string {
|
|
14
|
+
const common = [
|
|
15
|
+
'你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,',
|
|
16
|
+
'请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。',
|
|
17
|
+
'只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。',
|
|
18
|
+
'保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。',
|
|
19
|
+
'描述控制在 200-600 字左右。',
|
|
20
|
+
].join('')
|
|
21
|
+
const perMode: Record<AudioMode, string> = {
|
|
22
|
+
tts: '这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。',
|
|
23
|
+
music: '这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。',
|
|
24
|
+
sfx: '这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。',
|
|
25
|
+
voice_design: '这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。',
|
|
26
|
+
}
|
|
27
|
+
return common + perMode[mode]
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface PromptEnhanceDeps {
|
|
31
|
+
/** DSH 设置 seam(读 agent-default-model)。 */
|
|
32
|
+
settings: { describe(options?: { redactSecrets?: boolean }): Array<{ ns: unknown; value?: unknown }> }
|
|
33
|
+
/** 宿主 LLM 运行时访问器(延迟读取,调用时才获取)。 */
|
|
34
|
+
llm?: () => unknown
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
|
|
38
|
+
export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string, mode: AudioMode): Promise<string> {
|
|
39
|
+
const text = prompt.trim()
|
|
40
|
+
if (text === '') throw new AudioGenError('提示词为空,无法增强', 'enhance-empty-prompt')
|
|
41
|
+
const descriptor = (deps.settings.describe({ redactSecrets: true }) ?? []).find(candidate => String(candidate.ns) === 'agent-default-model')
|
|
42
|
+
const value = (descriptor?.value ?? {}) as { provider?: unknown; model?: unknown }
|
|
43
|
+
const provider = typeof value.provider === 'string' && value.provider.trim() !== '' ? value.provider.trim() : ''
|
|
44
|
+
const model = typeof value.model === 'string' && value.model.trim() !== '' ? value.model.trim() : ''
|
|
45
|
+
if (provider === '' || model === '') {
|
|
46
|
+
throw new AudioGenError('未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型', 'no-default-model')
|
|
47
|
+
}
|
|
48
|
+
const runtime = deps.llm?.() as { stream?: (options: unknown) => AsyncIterable<unknown> } | undefined
|
|
49
|
+
if (runtime === undefined || runtime.stream === undefined) {
|
|
50
|
+
throw new AudioGenError('宿主 LLM 服务不可用(ctx.llm 未注册)', 'llm-unavailable')
|
|
51
|
+
}
|
|
52
|
+
let output = ''
|
|
53
|
+
for await (const chunk of runtime.stream({
|
|
54
|
+
provider,
|
|
55
|
+
model,
|
|
56
|
+
messages: [{ role: 'user', content: text }],
|
|
57
|
+
system: instructionsFor(mode),
|
|
58
|
+
temperature: 0.7,
|
|
59
|
+
maxTokens: 1200,
|
|
60
|
+
})) {
|
|
61
|
+
const record = chunk as { type?: string; text?: string }
|
|
62
|
+
if (record.type === 'text-delta' && typeof record.text === 'string') output += record.text
|
|
63
|
+
}
|
|
64
|
+
return stripFences(output.trim())
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** 去掉模型可能包裹的 ``` 代码围栏。 */
|
|
68
|
+
function stripFences(value: string): string {
|
|
69
|
+
if (value === '') return value
|
|
70
|
+
const withoutFence = value.replace(/^```[a-zA-Z]*\s*\n?/, '').replace(/\n?```\s*$/, '')
|
|
71
|
+
return withoutFence.trim()
|
|
72
|
+
}
|
package/src/protocol.ts
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
|
|
9
9
|
|
|
10
10
|
/** Published package version shared by the host updater and the client UI. */
|
|
11
|
-
export const PLUGIN_VERSION = '0.4.
|
|
11
|
+
export const PLUGIN_VERSION = '0.4.5'
|
|
12
12
|
|
|
13
13
|
/** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
|
|
14
14
|
export const SETTINGS_API = {
|
|
@@ -24,6 +24,9 @@ export const TASK_API = {
|
|
|
24
24
|
cancel: '/api/dsh-audiogen/task/cancel',
|
|
25
25
|
} as const
|
|
26
26
|
|
|
27
|
+
/** Loopback-only prompt enhancement route (uses the agent's default model). */
|
|
28
|
+
export const ENHANCE_API = '/api/dsh-audiogen/prompt/enhance' as const
|
|
29
|
+
|
|
27
30
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
28
31
|
export const PRESETS_API = '/api/dsh-audiogen/presets' as const
|
|
29
32
|
|
package/src/routes.ts
CHANGED
|
@@ -16,7 +16,7 @@ import { discoverAudioModels } from './audio-models.ts'
|
|
|
16
16
|
import { AUDIO_PRESETS } from './audio-presets.ts'
|
|
17
17
|
import { appendHistory, clearHistory, listHistory, readAudioFile, removeHistory, saveAudioFile, listLibrary, saveToLibrary, updateLibraryEntry, removeLibraryEntries, readLibraryFile } from './audio-store.ts'
|
|
18
18
|
import {
|
|
19
|
-
AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
|
|
19
|
+
AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
|
|
20
20
|
LIBRARY_TYPES,
|
|
21
21
|
type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput, type LibraryAudioInput, type LibraryProvenance, type LibraryType,
|
|
22
22
|
} from './protocol.ts'
|
|
@@ -47,6 +47,8 @@ export interface AudiogenRoutesDeps {
|
|
|
47
47
|
autoSave: () => boolean
|
|
48
48
|
/** Global upstream concurrency gate (maxConcurrentGenerations). */
|
|
49
49
|
budget: GenerationBudget
|
|
50
|
+
/** 提示词增强:调用 Agent 默认模型,返回增强后的文本。 */
|
|
51
|
+
enhance: (prompt: string, mode: GenerateAudioRequest['mode']) => Promise<string>
|
|
50
52
|
}
|
|
51
53
|
|
|
52
54
|
function isLoopbackRequest(request: IncomingMessage): boolean {
|
|
@@ -528,6 +530,27 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
|
|
|
528
530
|
writeJson(res, 200, { ok: true, aborted: controllers !== undefined ? controllers.size : 0 })
|
|
529
531
|
},
|
|
530
532
|
},
|
|
533
|
+
// ------------------------------------------------------- prompt enhance
|
|
534
|
+
{
|
|
535
|
+
kind: 'exact',
|
|
536
|
+
path: ENHANCE_API,
|
|
537
|
+
handler: async (req, res) => {
|
|
538
|
+
if (!guard(req, res, 'POST')) return
|
|
539
|
+
const body = await readJsonBody(req)
|
|
540
|
+
const prompt = typeof body?.prompt === 'string' ? body.prompt.trim() : ''
|
|
541
|
+
if (prompt === '') {
|
|
542
|
+
writeJson(res, 200, { ok: false, code: 'bad-request', message: 'prompt is required' })
|
|
543
|
+
return
|
|
544
|
+
}
|
|
545
|
+
const mode = body?.mode === 'music' ? 'music' : body?.mode === 'sfx' ? 'sfx' : body?.mode === 'voice_design' ? 'voice_design' : 'tts'
|
|
546
|
+
try {
|
|
547
|
+
const enhanced = await deps.enhance(prompt, mode)
|
|
548
|
+
writeJson(res, 200, { ok: true, enhanced })
|
|
549
|
+
} catch (error) {
|
|
550
|
+
writeJson(res, 200, { ok: false, code: 'enhance-failed', message: messageOf(error) })
|
|
551
|
+
}
|
|
552
|
+
},
|
|
553
|
+
},
|
|
531
554
|
// ----------------------------------------------------------- audio file
|
|
532
555
|
{
|
|
533
556
|
kind: 'prefix',
|